WolfDavid commited on
Commit
f4874f3
·
1 Parent(s): a960e15

feat(01-08): wire the turn loop - text and voice turns, thinking, replay, slower, latency

Browse files

- avatar/turn-loop.js: dispatchTurn engages the thinking pose before the first await,
calls the host bridge with one JSON argument, speaks the directive, stamps
lastTurnMs at speech-start (where thinking clears) and publishes lastStageTimings,
lastSubtitle, turnCount, replayCount and lastReplayMs; replay stays networkless;
requestSlower re-dispatches the last subtitle at speedScale 0.75; the deferred stubs
are gone. Implemented once - avatar.js and avatar-iframe.js changed by zero lines.
- avatar/host.js (new): the host glue, loaded by the shared boot template for both
transports. Binds Enter/Send/Say hello/Replay/Slower/push-to-talk to the facade,
echoes typed text before the round trip, renders status, transcript, the latency
breakdown and the ASR tier badge, disables the controls while thinking or speaking
and re-arms them 200 ms after speech-end. Skips the Enter that ends an IME composition.
- avatar/vrm-stage.js publishes headPitch and relaxedValue so the thinking pose is a
rendered number, not a flag.
- tests: standalone thinking-pose test; parity suite now drives a real turn, the
thinking transition, a request-counted replay and a slower re-read under BOTH
transports and retires the deferred-stub guard; seam guards for the ordering, the
networkless replay, the published numbers and the host glue; the remount test drives
the wave-5 controls.

The turn loop's getDebug() returns the facade's live merged object, so probes read
scalars out the moment they sample.

avatar/host.js ADDED
@@ -0,0 +1,235 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ // avatar/host.js
2
+ //
3
+ // THE HOST GLUE: binds the page's controls to window.Avatar and renders what the avatar
4
+ // reports. It is the only module in avatar/ that knows the controls' element ids, and it
5
+ // knows NOTHING about which transport booted - it is handed the facade object and talks
6
+ // to nothing else. Both transports load it from the same boot template, so the controls
7
+ // behave identically under either by construction.
8
+ //
9
+ // It implements no turn behaviour. Every control below is one call into the facade;
10
+ // the behaviour lives in avatar/turn-loop.js. tests/test_transport_seam.py enforces
11
+ // both halves of that: this file is imported by no transport and defines no turn method.
12
+ //
13
+ // DOM writes go to inner elements this project owns (#status-text, #transcript-text,
14
+ // #latency-text, #asr-tier-text) rather than to the host's wrapper elements, so a
15
+ // re-render of a wrapper cannot delete a line, and user-supplied text is always written
16
+ // through textContent, never innerHTML.
17
+
18
+ /** Matches REARM_TAIL_MS in mic.js: the controls re-enable when the mic may re-arm. */
19
+ const REENABLE_AFTER_SPEECH_MS = 200;
20
+
21
+ /** Enter finishes a Japanese IME composition before it submits; this skips that Enter. */
22
+ function isComposing(event) {
23
+ return event.isComposing || event.keyCode === 229;
24
+ }
25
+
26
+ /**
27
+ * @param {object} avatar the facade window.Avatar
28
+ * @param {Document} [doc]
29
+ * @returns {boolean} whether the bindings were installed by this call
30
+ */
31
+ export function bindHost(avatar, doc = document) {
32
+ if (!avatar || typeof avatar.on !== 'function') return false;
33
+ // boot() is re-entrant and returns the live object; the bindings must not double up.
34
+ if (doc.__avatarHostBound) return false;
35
+ doc.__avatarHostBound = true;
36
+
37
+ const byId = (id) => doc.getElementById(id);
38
+ const text = (id, value) => {
39
+ const el = byId(id);
40
+ if (el) el.textContent = value;
41
+ };
42
+ const status = (value) => text('status-text', value);
43
+
44
+ const textarea = () => doc.querySelector('#text-input textarea, #text-input input');
45
+ const controls = {
46
+ ptt: byId('ptt-button'),
47
+ hello: byId('hello-button'),
48
+ send: byId('send-button'),
49
+ replay: byId('replay-button'),
50
+ slower: byId('slower-button'),
51
+ };
52
+
53
+ let spoken = false; // whether anything has been said yet, for replay/slower
54
+ let busy = false;
55
+ let reenableTimer = null;
56
+
57
+ function applyEnabled() {
58
+ for (const [name, el] of Object.entries(controls)) {
59
+ if (!el) continue;
60
+ const needsSpeech = name === 'replay' || name === 'slower';
61
+ el.disabled = busy || (needsSpeech && !spoken);
62
+ }
63
+ }
64
+
65
+ function setBusy(value) {
66
+ busy = !!value;
67
+ if (reenableTimer) {
68
+ clearTimeout(reenableTimer);
69
+ reenableTimer = null;
70
+ }
71
+ applyEnabled();
72
+ }
73
+
74
+ function transcriptLine(who, value) {
75
+ const el = byId('transcript-text');
76
+ if (!el) return;
77
+ const line = doc.createElement('div');
78
+ line.className = `turn turn-${who}`;
79
+ const label = doc.createElement('span');
80
+ label.className = 'who';
81
+ label.textContent = who === 'you' ? 'You: ' : who === 'slower' ? 'Avatar (slower): ' : 'Avatar: ';
82
+ const body = doc.createElement('span');
83
+ body.className = 'said';
84
+ body.textContent = value;
85
+ line.append(label, body);
86
+ el.append(line);
87
+ el.scrollTop = el.scrollHeight;
88
+ }
89
+
90
+ function renderLatency({ lastTurnMs, timings }) {
91
+ const t = timings || {};
92
+ const ms = (key) => (typeof t[key] === 'number' ? Math.round(t[key]) : '—');
93
+ text(
94
+ 'latency-text',
95
+ `dispatch→speech: ${lastTurnMs} ms (server: query ${ms('audio_query_ms')} / ` +
96
+ `synth ${ms('synthesis_ms')} / timeline ${ms('timeline_ms')} / encode ${ms('encode_ms')})`
97
+ );
98
+ }
99
+
100
+ function renderTier({ tier, model, dtype }) {
101
+ const shortModel = String(model || '').split('/').pop();
102
+ const label = tier === 'webgpu' ? 'WebGPU' : tier === 'wasm' ? 'WASM' : String(tier);
103
+ text('asr-tier-text', `ASR: ${label} · ${shortModel} ${dtype || ''}`.trim());
104
+ }
105
+
106
+ function report(err) {
107
+ status(`error: ${String(err?.message ?? err)}`);
108
+ }
109
+
110
+ function dispatch(value, opts) {
111
+ setBusy(true);
112
+ status('thinking…');
113
+ avatar.dispatchTurn(value, opts).catch(report);
114
+ }
115
+
116
+ function submitText() {
117
+ const el = textarea();
118
+ const value = el ? el.value.trim() : '';
119
+ if (!value) {
120
+ status('type something in Japanese first');
121
+ return;
122
+ }
123
+ // Echo before the round trip, so the visitor sees their words the instant they send.
124
+ transcriptLine('you', value);
125
+ if (el) {
126
+ el.value = '';
127
+ // The host's textbox mirrors its value from input events; a bare .value write
128
+ // would leave the host believing the old text is still there.
129
+ el.dispatchEvent(new Event('input', { bubbles: true }));
130
+ }
131
+ dispatch(value);
132
+ }
133
+
134
+ // ------------------------------------------------------------------- avatar -> page
135
+ avatar.on('listening', ({ active } = {}) => status(active ? 'listening…' : 'transcribing…'));
136
+ avatar.on('transcript', ({ text: heard } = {}) => {
137
+ if (!heard) return;
138
+ transcriptLine('you', heard);
139
+ dispatch(heard);
140
+ });
141
+ avatar.on('turn-start', () => {
142
+ setBusy(true);
143
+ status('thinking…');
144
+ });
145
+ avatar.on('turn', ({ subtitle, speed, greeting } = {}) => {
146
+ if (subtitle && (greeting || speed < 1)) transcriptLine(speed < 1 ? 'slower' : 'avatar', subtitle);
147
+ spoken = true;
148
+ });
149
+ avatar.on('speech-start', () => {
150
+ setBusy(true);
151
+ status('speaking…');
152
+ });
153
+ avatar.on('speech-end', () => {
154
+ status('ready');
155
+ reenableTimer = setTimeout(() => setBusy(false), REENABLE_AFTER_SPEECH_MS);
156
+ });
157
+ avatar.on('latency', renderLatency);
158
+ avatar.on('asr-tier', renderTier);
159
+ avatar.on('error', ({ message, where } = {}) => {
160
+ setBusy(false);
161
+ status(`error (${where || 'avatar'}): ${message}`);
162
+ });
163
+
164
+ // ------------------------------------------------------------------- page -> avatar
165
+ if (controls.send) controls.send.addEventListener('click', submitText);
166
+
167
+ const input = textarea();
168
+ if (input) {
169
+ input.addEventListener('keydown', (event) => {
170
+ if (event.key !== 'Enter' || event.shiftKey || isComposing(event)) return;
171
+ event.preventDefault();
172
+ submitText();
173
+ });
174
+ }
175
+
176
+ if (controls.hello) {
177
+ controls.hello.addEventListener('click', () => dispatch('', { greeting: true }));
178
+ }
179
+
180
+ if (controls.replay) {
181
+ controls.replay.addEventListener('click', () => {
182
+ setBusy(true);
183
+ avatar.replay().catch(report);
184
+ });
185
+ }
186
+
187
+ if (controls.slower) {
188
+ controls.slower.addEventListener('click', () => {
189
+ setBusy(true);
190
+ status('thinking…');
191
+ avatar.requestSlower().catch(report);
192
+ });
193
+ }
194
+
195
+ if (controls.ptt) {
196
+ const ptt = controls.ptt;
197
+ ptt.style.touchAction = 'none';
198
+ ptt.addEventListener('contextmenu', (event) => event.preventDefault());
199
+ ptt.addEventListener('pointerdown', (event) => {
200
+ event.preventDefault();
201
+ if (ptt.setPointerCapture) {
202
+ try {
203
+ ptt.setPointerCapture(event.pointerId);
204
+ } catch {
205
+ /* capture is a nicety; release still arrives on the button */
206
+ }
207
+ }
208
+ avatar
209
+ .startListening()
210
+ .then(async (started) => {
211
+ if (started) return;
212
+ const d = await avatar.getDebug();
213
+ status(`microphone did not open (${d.micLastRejectReason || 'refused'})`);
214
+ })
215
+ .catch(report);
216
+ });
217
+ const release = () => {
218
+ avatar
219
+ .stopListening()
220
+ .then(async (heard) => {
221
+ if (heard) return; // the 'transcript' event has already dispatched the turn
222
+ const d = await avatar.getDebug();
223
+ const reason = d.micLastRejectReason ? ` (${d.micLastRejectReason})` : '';
224
+ status(`didn't catch that${reason} - hold the button and speak`);
225
+ })
226
+ .catch(report);
227
+ };
228
+ ptt.addEventListener('pointerup', release);
229
+ ptt.addEventListener('pointercancel', release);
230
+ }
231
+
232
+ applyEnabled();
233
+ status('ready - hold the button and speak, or type Japanese below');
234
+ return true;
235
+ }
avatar/turn-loop.js CHANGED
@@ -14,20 +14,19 @@
14
  // two transport files needed zero lines of change to gain push-to-talk, which is the
15
  // strongest possible form of the guarantee the seam exists to give. Both are still
16
  // injectable through the factory so a harness can substitute a different model.
 
 
 
 
 
 
 
17
 
18
  import { createAsr } from './asr.js';
19
  import { createMic, isHallucination, REJECT } from './mic.js';
20
 
21
- /**
22
- * A deferred method. Calling one now fails loudly and names the plan that fills it in,
23
- * which is the difference between a placeholder and an accidental no-op. The parity
24
- * guard is therefore meaningful from this wave rather than only after the last one.
25
- */
26
- function notWiredYet(name, plan) {
27
- return () => {
28
- throw new Error(`Avatar.${name}() is not wired yet - plan ${plan} implements it`);
29
- };
30
- }
31
 
32
  /**
33
  * @param {object} opts
@@ -57,30 +56,57 @@ export function createTurnLoop({
57
  const state = {
58
  thinking: false,
59
  listening: false,
 
60
  lastTurnId: null,
61
  replayCount: 0,
 
 
 
 
 
 
 
 
 
62
  asrTier: null,
63
  asrModel: null,
64
  lastTranscript: null,
65
  micRejectedCount: 0,
 
66
  };
67
 
68
- let speaking = false;
 
 
 
 
 
 
 
69
 
70
- function setListeningState(on) {
 
 
 
 
 
 
 
 
71
  state.listening = on;
72
  stagePort.setListening(on);
 
73
  }
74
 
75
  const micInstance =
76
  mic ||
77
  createMic({
78
  emit,
79
- onListening: setListeningState,
80
  // Push-to-talk exists to make acoustic feedback impossible, so the mic refuses to
81
  // open while the avatar is thinking or speaking. mic.js adds the 200 ms tail after
82
  // speech-end on top of this.
83
- isBusy: () => state.thinking || speaking,
84
  ...micOptions,
85
  });
86
 
@@ -94,9 +120,34 @@ export function createTurnLoop({
94
  */
95
  function observe(name, data) {
96
  if (name === 'speech-start') {
97
- speaking = true;
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
98
  } else if (name === 'speech-end') {
99
- speaking = false;
100
  micInstance.noteSpeechEnd();
101
  } else if (name === 'asr-tier' && data) {
102
  state.asrTier = data.tier ?? null;
@@ -104,51 +155,163 @@ export function createTurnLoop({
104
  }
105
  }
106
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
107
  return {
108
  state,
109
  getServer,
110
  observe,
111
  mic: micInstance,
112
  asr: asrInstance,
113
-
114
- setThinking(value) {
115
- const on = !!value;
116
- state.thinking = on;
117
- stagePort.setThinking(on);
118
- return on;
119
- },
120
-
121
- setListening(value) {
122
- const on = !!value;
123
- state.listening = on;
124
- stagePort.setListening(on);
125
- return on;
126
- },
127
 
128
  /**
129
  * Re-play the cached directive. There is deliberately no network access of any
130
- * kind in here: plan 01-09's test_replay asserts zero requests with a browser
131
- * request listener, and a cache miss must fail rather than quietly re-download.
 
132
  */
133
  async replay() {
 
 
 
134
  state.replayCount += 1;
 
 
 
135
  try {
136
  return await stagePort.replayCached();
137
  } catch (err) {
138
- emit('error', { message: String(err?.message ?? err), where: 'replay' });
139
- throw err;
 
 
 
 
 
 
 
 
 
 
 
 
140
  }
 
141
  },
142
 
143
  /**
144
- * pointerdown on the push-to-talk control. Plan 01-08 wires the control itself;
145
- * the behaviour is here so both transports get it from one implementation.
146
  *
147
  * @returns {Promise<boolean>} whether capture actually started
148
  */
149
  async startListening() {
150
  const started = await micInstance.start();
151
- if (!started) state.micRejectedCount = micInstance.__debug.rejectedCount;
 
 
 
152
  return started;
153
  },
154
 
@@ -163,13 +326,14 @@ export function createTurnLoop({
163
  async stopListening() {
164
  const utterance = await micInstance.stop();
165
  state.micRejectedCount = micInstance.__debug.rejectedCount;
 
166
  if (!utterance.ok) return null;
167
 
168
  let result;
169
  try {
170
  result = await asrInstance.transcribe(utterance.samples, utterance.sampleRate);
171
  } catch (err) {
172
- emit('error', { message: String(err?.message ?? err), where: 'stopListening' });
173
  return null;
174
  }
175
 
@@ -183,6 +347,7 @@ export function createTurnLoop({
183
  result.text ? REJECT.HALLUCINATION : REJECT.NO_AUDIO
184
  );
185
  state.micRejectedCount = micInstance.__debug.rejectedCount;
 
186
  return null;
187
  }
188
 
@@ -195,8 +360,5 @@ export function createTurnLoop({
195
  });
196
  return result.text;
197
  },
198
-
199
- dispatchTurn: notWiredYet('dispatchTurn', '01-08'),
200
- requestSlower: notWiredYet('requestSlower', '01-08'),
201
  };
202
  }
 
14
  // two transport files needed zero lines of change to gain push-to-talk, which is the
15
  // strongest possible form of the guarantee the seam exists to give. Both are still
16
  // injectable through the factory so a harness can substitute a different model.
17
+ //
18
+ // The turn itself landed in wave 5, in this file and nowhere else, so the same holds:
19
+ // dispatchTurn, replay and requestSlower work under both transports because there is
20
+ // exactly one implementation of each. The bridge object getServer() returns exposes
21
+ // the host-registered functions (turn, greeting) as async methods; this module calls
22
+ // them with ONE argument each, because the host's bridge packs multiple arguments into
23
+ // a list and the far side would receive that list as a single positional.
24
 
25
  import { createAsr } from './asr.js';
26
  import { createMic, isHallucination, REJECT } from './mic.js';
27
 
28
+ /** VOICEVOX speedScale for the "Slower" re-read. Divides every phoneme length. */
29
+ export const SLOWER_SPEED = 0.75;
 
 
 
 
 
 
 
 
30
 
31
  /**
32
  * @param {object} opts
 
56
  const state = {
57
  thinking: false,
58
  listening: false,
59
+ speaking: false,
60
  lastTurnId: null,
61
  replayCount: 0,
62
+ turnCount: 0,
63
+ // The number that matters: dispatch (Enter, click or mic release) to the first
64
+ // scheduled audio sample, in milliseconds, for the most recent turn.
65
+ lastTurnMs: null,
66
+ lastReplayMs: null,
67
+ lastStageTimings: null,
68
+ lastSubtitle: null,
69
+ lastSpeed: null,
70
+ lastError: null,
71
  asrTier: null,
72
  asrModel: null,
73
  lastTranscript: null,
74
  micRejectedCount: 0,
75
+ micLastRejectReason: null,
76
  };
77
 
78
+ // Set when a turn or a replay has been dispatched and its speech-start has not yet
79
+ // been observed. speech-start is the event that closes the "thinking" window and
80
+ // stamps lastTurnMs, so it is measured where the event arrives - the same place under
81
+ // both transports - rather than guessed at from the speak() promise.
82
+ let pendingDispatchAt = null;
83
+ let pendingReplayAt = null;
84
+
85
+ const now = () => performance.now();
86
 
87
+ function setThinking(value) {
88
+ const on = !!value;
89
+ state.thinking = on;
90
+ stagePort.setThinking(on);
91
+ return on;
92
+ }
93
+
94
+ function setListening(value) {
95
+ const on = !!value;
96
  state.listening = on;
97
  stagePort.setListening(on);
98
+ return on;
99
  }
100
 
101
  const micInstance =
102
  mic ||
103
  createMic({
104
  emit,
105
+ onListening: setListening,
106
  // Push-to-talk exists to make acoustic feedback impossible, so the mic refuses to
107
  // open while the avatar is thinking or speaking. mic.js adds the 200 ms tail after
108
  // speech-end on top of this.
109
+ isBusy: () => state.thinking || state.speaking,
110
  ...micOptions,
111
  });
112
 
 
120
  */
121
  function observe(name, data) {
122
  if (name === 'speech-start') {
123
+ state.speaking = true;
124
+ if (pendingDispatchAt !== null) {
125
+ state.lastTurnMs = Math.round(now() - pendingDispatchAt);
126
+ pendingDispatchAt = null;
127
+ performance.mark('turn:speech-start');
128
+ try {
129
+ performance.measure('turn:dispatch-to-speech', 'turn:dispatch', 'turn:speech-start');
130
+ } catch {
131
+ /* a mark was cleared; the number is already in lastTurnMs */
132
+ }
133
+ // The thinking pose clears HERE, at speech-start, not when the response arrives:
134
+ // decode and scheduling still sit between the two, and the face must not go idle
135
+ // while the visitor is still waiting to hear something.
136
+ setThinking(false);
137
+ emit('latency', {
138
+ turnId: state.lastTurnId,
139
+ lastTurnMs: state.lastTurnMs,
140
+ timings: state.lastStageTimings,
141
+ speed: state.lastSpeed,
142
+ });
143
+ }
144
+ if (pendingReplayAt !== null) {
145
+ state.lastReplayMs = Math.round(now() - pendingReplayAt);
146
+ pendingReplayAt = null;
147
+ performance.mark('replay:speech-start');
148
+ }
149
  } else if (name === 'speech-end') {
150
+ state.speaking = false;
151
  micInstance.noteSpeechEnd();
152
  } else if (name === 'asr-tier' && data) {
153
  state.asrTier = data.tier ?? null;
 
155
  }
156
  }
157
 
158
+ function busyReason() {
159
+ if (state.thinking) return 'the avatar is still thinking about the last turn';
160
+ if (state.speaking) return 'the avatar is still speaking';
161
+ return null;
162
+ }
163
+
164
+ function fail(where, err) {
165
+ const message = String(err?.message ?? err);
166
+ state.lastError = message;
167
+ emit('error', { message, where });
168
+ return err instanceof Error ? err : new Error(message);
169
+ }
170
+
171
+ /**
172
+ * One turn: text in, speech out. Resolves at speech-end with a summary of the turn.
173
+ *
174
+ * Order matters and is asserted statically: the thinking pose engages BEFORE the
175
+ * first await, because switching the pose when the response arrives would forfeit
176
+ * the entire latency the pose exists to cover.
177
+ *
178
+ * @param {string} text what the avatar should say back
179
+ * @param {object} [opts]
180
+ * @param {number} [opts.speed=1.0] VOICEVOX speedScale; SLOWER_SPEED for the re-read
181
+ * @param {boolean} [opts.greeting] ignore text and speak the server's fixed greeting
182
+ */
183
+ async function dispatchTurn(text, { speed = 1.0, greeting = false } = {}) {
184
+ const busy = busyReason();
185
+ if (busy) throw fail('dispatchTurn', new Error(busy));
186
+
187
+ const bridge = getServer();
188
+ if (!bridge || typeof bridge.turn !== 'function') {
189
+ throw fail(
190
+ 'dispatchTurn',
191
+ new Error('no host bridge: the standalone stage has nothing to synthesise with')
192
+ );
193
+ }
194
+
195
+ performance.mark('turn:dispatch');
196
+ const dispatchedAt = now();
197
+ pendingDispatchAt = dispatchedAt;
198
+ setThinking(true);
199
+ emit('turn-start', { text: greeting ? null : text, speed, greeting });
200
+
201
+ try {
202
+ const directive = greeting
203
+ ? await bridge.greeting()
204
+ : await bridge.turn({ text: String(text ?? ''), speed });
205
+ performance.mark('turn:response');
206
+ const responseMs = Math.round(now() - dispatchedAt);
207
+
208
+ // The host's client swallows an HTTP error into `undefined`, and the far side
209
+ // answers a bad request with {error} rather than raising, so both are checked.
210
+ if (directive === undefined || directive === null) {
211
+ throw new Error('the host returned nothing for this turn - see its log');
212
+ }
213
+ if (directive.error) throw new Error(directive.error);
214
+ if (!directive.audio_url || !Array.isArray(directive.timeline)) {
215
+ throw new Error('the host returned a directive with no audio or no timeline');
216
+ }
217
+
218
+ state.turnCount += 1;
219
+ state.lastTurnId = directive.turn_id ?? null;
220
+ state.lastSubtitle = directive.subtitle ?? null;
221
+ state.lastSpeed = directive.speed ?? speed;
222
+ state.lastStageTimings = directive.timings ?? null;
223
+ state.lastError = null;
224
+ emit('turn', {
225
+ turnId: state.lastTurnId,
226
+ subtitle: state.lastSubtitle,
227
+ speed: state.lastSpeed,
228
+ timings: state.lastStageTimings,
229
+ responseMs,
230
+ greeting,
231
+ });
232
+
233
+ // Resolves at speech-end. speech-start arrives through observe() on the way.
234
+ const played = await stagePort.speak({
235
+ audioUrl: directive.audio_url,
236
+ timeline: directive.timeline,
237
+ subtitle: directive.subtitle,
238
+ expression: directive.expression,
239
+ turnId: directive.turn_id,
240
+ });
241
+
242
+ return {
243
+ turnId: state.lastTurnId,
244
+ subtitle: state.lastSubtitle,
245
+ speed: state.lastSpeed,
246
+ timings: state.lastStageTimings,
247
+ responseMs,
248
+ lastTurnMs: state.lastTurnMs,
249
+ duration: played?.duration ?? null,
250
+ };
251
+ } catch (err) {
252
+ pendingDispatchAt = null;
253
+ setThinking(false);
254
+ throw fail('dispatchTurn', err);
255
+ }
256
+ }
257
+
258
  return {
259
  state,
260
  getServer,
261
  observe,
262
  mic: micInstance,
263
  asr: asrInstance,
264
+ setThinking,
265
+ setListening,
266
+ dispatchTurn,
 
 
 
 
 
 
 
 
 
 
 
267
 
268
  /**
269
  * Re-play the cached directive. There is deliberately no network access of any
270
+ * kind in here: the deployed suite asserts zero requests with a browser request
271
+ * listener, and a cache miss must fail rather than quietly re-download. The stage
272
+ * keeps the decoded AudioBuffer and the timeline it last spoke.
273
  */
274
  async replay() {
275
+ const busy = busyReason();
276
+ if (busy) throw fail('replay', new Error(busy));
277
+ if (state.turnCount === 0) throw fail('replay', new Error('nothing has been said yet'));
278
  state.replayCount += 1;
279
+ performance.mark('replay:dispatch');
280
+ pendingReplayAt = now();
281
+ emit('replay', { turnId: state.lastTurnId, subtitle: state.lastSubtitle });
282
  try {
283
  return await stagePort.replayCached();
284
  } catch (err) {
285
+ pendingReplayAt = null;
286
+ throw fail('replay', err);
287
+ }
288
+ },
289
+
290
+ /**
291
+ * Re-synthesise the last utterance at SLOWER_SPEED. This IS a host round trip and
292
+ * must be: the timeline has to be rebuilt from the re-synthesised query, because
293
+ * speedScale divides every phoneme and a timeline scaled here would drift by exactly
294
+ * the speed ratio against the new audio.
295
+ */
296
+ async requestSlower() {
297
+ if (!state.lastSubtitle) {
298
+ throw fail('requestSlower', new Error('nothing to slow down yet - say something first'));
299
  }
300
+ return dispatchTurn(state.lastSubtitle, { speed: SLOWER_SPEED });
301
  },
302
 
303
  /**
304
+ * pointerdown on the push-to-talk control. The host binds the control; the behaviour
305
+ * is here so both transports get it from one implementation.
306
  *
307
  * @returns {Promise<boolean>} whether capture actually started
308
  */
309
  async startListening() {
310
  const started = await micInstance.start();
311
+ if (!started) {
312
+ state.micRejectedCount = micInstance.__debug.rejectedCount;
313
+ state.micLastRejectReason = micInstance.__debug.lastRejectReason;
314
+ }
315
  return started;
316
  },
317
 
 
326
  async stopListening() {
327
  const utterance = await micInstance.stop();
328
  state.micRejectedCount = micInstance.__debug.rejectedCount;
329
+ state.micLastRejectReason = micInstance.__debug.lastRejectReason;
330
  if (!utterance.ok) return null;
331
 
332
  let result;
333
  try {
334
  result = await asrInstance.transcribe(utterance.samples, utterance.sampleRate);
335
  } catch (err) {
336
+ fail('stopListening', err);
337
  return null;
338
  }
339
 
 
347
  result.text ? REJECT.HALLUCINATION : REJECT.NO_AUDIO
348
  );
349
  state.micRejectedCount = micInstance.__debug.rejectedCount;
350
+ state.micLastRejectReason = micInstance.__debug.lastRejectReason;
351
  return null;
352
  }
353
 
 
360
  });
361
  return result.text;
362
  },
 
 
 
363
  };
364
  }
avatar/vrm-stage.js CHANGED
@@ -150,6 +150,13 @@ export async function mountStage(canvasEl, vrmUrl, emit = () => {}) {
150
  // -1 is straight down, 0 is the T-pose, +1 is straight up. Tests assert on this so
151
  // a regression to the T-pose fails a number instead of needing an eyeball.
152
  armDown: { left: 0, right: 0 },
 
 
 
 
 
 
 
153
  clockOffset: 0,
154
  thinking: false,
155
  listening: false,
@@ -191,7 +198,8 @@ export async function mountStage(canvasEl, vrmUrl, emit = () => {}) {
191
  };
192
  const shoulderPos = new THREE.Vector3();
193
  const elbowPos = new THREE.Vector3();
194
- function measureArms() {
 
195
  for (const side of ['left', 'right']) {
196
  const [upper, lower] = armPairs[side];
197
  if (!upper || !lower) continue;
@@ -201,6 +209,14 @@ export async function mountStage(canvasEl, vrmUrl, emit = () => {}) {
201
  const len = dir.length();
202
  debug.armDown[side] = len > 0 ? dir.y / len : 0;
203
  }
 
 
 
 
 
 
 
 
204
  }
205
 
206
  const clock = new THREE.Clock();
@@ -267,7 +283,7 @@ export async function mountStage(canvasEl, vrmUrl, emit = () => {}) {
267
  if (onTick) onTick(dt);
268
  vrm.update(dt); // MUST run after expression values are set, every frame
269
  renderer.render(scene, camera);
270
- measureArms();
271
  });
272
 
273
  const ro = new ResizeObserver(() => {
 
150
  // -1 is straight down, 0 is the T-pose, +1 is straight up. Tests assert on this so
151
  // a regression to the T-pose fails a number instead of needing an eyeball.
152
  armDown: { left: 0, right: 0 },
153
+ // The thinking pose, as a viewer would see it: how far the head looks down (the
154
+ // world-space forward direction's downward component, ~sin THINK_TILT while
155
+ // thinking, ~0 otherwise, +/- the breath) and the 'relaxed' expression weight.
156
+ // Published so the pose is a number a test can assert, not a flag that says it was
157
+ // requested - the arms taught this project that those are different things.
158
+ headPitch: 0,
159
+ relaxedValue: 0,
160
  clockOffset: 0,
161
  thinking: false,
162
  listening: false,
 
198
  };
199
  const shoulderPos = new THREE.Vector3();
200
  const elbowPos = new THREE.Vector3();
201
+ const headForward = new THREE.Vector3();
202
+ function measurePose() {
203
  for (const side of ['left', 'right']) {
204
  const [upper, lower] = armPairs[side];
205
  if (!upper || !lower) continue;
 
209
  const len = dir.length();
210
  debug.armDown[side] = len > 0 ? dir.y / len : 0;
211
  }
212
+ // The normalized head bone is world-aligned at rest (+Z forward for every VRM), and
213
+ // three-vrm copies it onto the raw bone every update, so its world forward after a
214
+ // render is the direction the rendered face points. A positive pitch looks down.
215
+ if (head) {
216
+ head.getWorldDirection(headForward);
217
+ debug.headPitch = -headForward.y;
218
+ }
219
+ debug.relaxedValue = vrm.expressionManager?.getValue('relaxed') ?? 0;
220
  }
221
 
222
  const clock = new THREE.Clock();
 
283
  if (onTick) onTick(dt);
284
  vrm.update(dt); // MUST run after expression values are set, every frame
285
  renderer.render(scene, camera);
286
+ measurePose();
287
  });
288
 
289
  const ro = new ResizeObserver(() => {
src/japanese_avatar/ui/avatar_component.py CHANGED
@@ -27,11 +27,12 @@ VRM_URL = "/gradio_api/file=avatar/assets/tutor.vrm"
27
  # The IIFE also escapes Gradio's own try/catch, so it carries its own .catch: a boot
28
  # failure must reach the console, because that console line is plan 01-05's verdict.
29
  #
30
- # The status-line writes live HERE, in the shared boot template, rather than inside
31
- # avatar.js. Both transports run this same string in the host document, so the loading
32
- # state is symmetric by construction - putting it in avatar.js would have given the
33
- # inline transport a status line and the iframe fallback none, which is exactly the
34
- # drift the seam exists to prevent.
 
35
  #
36
  # Note what is NOT awaited before the stage mounts: nothing on the Python side. The
37
  # VRM is client-side, so it paints and starts breathing on its own schedule. On a
@@ -45,8 +46,11 @@ _BOOT_JS = """
45
  if (el) el.textContent = text;
46
  }};
47
  const m = await import('/gradio_api/file=avatar/{module}');
48
- await m.boot(element, props, trigger, server);
49
- status('ready - the avatar is live. Voice conversation is not wired up yet.');
 
 
 
50
  watch('value', () => m.onDirective(props.value));
51
  }})().catch((err) => {{
52
  console.error('avatar boot failed:', err);
@@ -102,10 +106,10 @@ _STATUS_HTML = '<div id="status-text" class="status-line">waking up...</div>'
102
  class StatusLine(gr.HTML):
103
  """The one-line "what is the avatar doing" readout, addressable as #status-line.
104
 
105
- Plans 01-07 and 01-08 write listening / thinking / speaking states into it. It is
106
- a component rather than a bare string in app.py so the elem_id and the inner
107
  #status-text id are declared in exactly one place, next to the boot script that
108
- writes to them.
109
  """
110
 
111
  def __init__(self, **kwargs):
 
27
  # The IIFE also escapes Gradio's own try/catch, so it carries its own .catch: a boot
28
  # failure must reach the console, because that console line is plan 01-05's verdict.
29
  #
30
+ # The status-line writes and the control bindings live HERE (via avatar/host.js), in the
31
+ # shared boot template, rather than inside avatar.js. Both transports run this same
32
+ # string in the host document, so the loading state and every control are symmetric by
33
+ # construction - putting them in avatar.js would have given the inline transport a
34
+ # working page and the iframe fallback an inert one, which is exactly the drift the seam
35
+ # exists to prevent.
36
  #
37
  # Note what is NOT awaited before the stage mounts: nothing on the Python side. The
38
  # VRM is client-side, so it paints and starts breathing on its own schedule. On a
 
46
  if (el) el.textContent = text;
47
  }};
48
  const m = await import('/gradio_api/file=avatar/{module}');
49
+ const avatar = await m.boot(element, props, trigger, server);
50
+ // The host glue: binds the page's controls to the facade and renders what it
51
+ // reports. Loaded here, after boot, from the SAME template for both transports.
52
+ const host = await import('/gradio_api/file=avatar/host.js');
53
+ host.bindHost(avatar, document);
54
  watch('value', () => m.onDirective(props.value));
55
  }})().catch((err) => {{
56
  console.error('avatar boot failed:', err);
 
106
  class StatusLine(gr.HTML):
107
  """The one-line "what is the avatar doing" readout, addressable as #status-line.
108
 
109
+ avatar/host.js writes the listening / thinking / speaking states into it. It is a
110
+ component rather than a bare string in the layout so the elem_id and the inner
111
  #status-text id are declared in exactly one place, next to the boot script that
112
+ loads the module which writes to them.
113
  """
114
 
115
  def __init__(self, **kwargs):
tests/e2e/test_avatar_loop.py CHANGED
@@ -286,15 +286,20 @@ def test_no_remount(page, space_url, warm_space):
286
  before = read_debug(page)
287
  assert before["mountCount"] == 1, f"mountCount was already {before['mountCount']} at ready"
288
 
 
 
 
 
 
289
  for i in range(20):
290
  which = i % 3
291
  if which == 0:
292
- page.click("#send")
293
  elif which == 1:
294
- page.fill("#text-input textarea", f"こんにちは {i}")
295
- page.click("#send")
296
  else:
297
- page.locator("#slower input[type=checkbox]").click()
298
  page.wait_for_timeout(250)
299
 
300
  debug = read_debug(page)
 
286
  before = read_debug(page)
287
  assert before["mountCount"] == 1, f"mountCount was already {before['mountCount']} at ready"
288
 
289
+ # The wave-5 controls. Send with an empty box is refused in the browser; a filled
290
+ # box submitted with Enter is a real turn through server_functions (the button is
291
+ # disabled while the avatar thinks and speaks, and Playwright's actionability wait
292
+ # absorbs that); the About accordion is a Gradio component toggle. Between them the
293
+ # host re-renders the right-hand column and the stage must not notice.
294
  for i in range(20):
295
  which = i % 3
296
  if which == 0:
297
+ page.click("#send-button")
298
  elif which == 1:
299
+ page.fill("#text-input input", f"こんにちは {i}")
300
+ page.press("#text-input input", "Enter")
301
  else:
302
+ page.locator("#about-panel button").first.click()
303
  page.wait_for_timeout(250)
304
 
305
  debug = read_debug(page)
tests/e2e/test_facade_parity.py CHANGED
@@ -1,19 +1,29 @@
1
- """Mechanical proof that the two transports expose the same object.
2
 
3
  The static tests in tests/test_transport_seam.py can only prove that no transport
4
  *writes* window.Avatar. These boot the real Gradio app twice - once per
5
  AVATAR_TRANSPORT value - and compare the LIVE objects, which is the only way to catch
6
  a method that resolves in one transport and silently does not in the other.
7
 
8
- Marked slow but NOT deployed: they run entirely locally and must be green before
9
- plan 01-05 deploys anything.
 
 
 
 
 
10
  """
11
 
12
  from __future__ import annotations
13
 
14
  import pytest
15
 
16
- from tests.e2e.test_stage_standalone import ARM_DOWN_MAX, FIRST_FRAME_TIMEOUT_MS
 
 
 
 
 
17
  from tests.test_transport_seam import avatar_surface
18
 
19
  pytestmark = pytest.mark.slow
@@ -21,6 +31,15 @@ pytestmark = pytest.mark.slow
21
  TRANSPORTS = ("inline", "iframe")
22
  AVATAR_READY = "() => !!window.Avatar && !!window.Avatar.__debug && window.Avatar.__debug.ready"
23
  BOOT_TIMEOUT_MS = 120_000
 
 
 
 
 
 
 
 
 
24
 
25
  # The pose is measured after a render, and ready fires before the first one. Same
26
  # frame-rendered signal as the standalone suite, read through the facade because the
@@ -32,60 +51,191 @@ async () => {
32
  }
33
  """
34
 
35
- # Deliberately double-quoted and built from data: the surface list must come from
36
- # facade.js, never from a literal in this file.
37
- #
38
- # This tuple shrinks as waves land. Plan 01-07 implemented startListening/stopListening,
39
- # so they moved OUT of here and into WIRED_METHODS below; only plan 01-08's two remain.
40
- # The test itself is not deleted until the tuple is empty - a deferred stub that has
41
- # quietly become a no-op is exactly the drift this suite exists to catch.
42
- DEFERRED_PLAN = "01-08"
43
- DEFERRED_METHODS = ("dispatchTurn", "requestSlower")
44
-
45
- # Implemented and expected to work under BOTH transports.
46
- WIRED_PLAN = "01-07"
47
- WIRED_METHODS = ("startListening", "stopListening")
48
-
49
- CALL_DEFERRED = """
50
- async (name) => {
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
51
  try {
52
- await window.Avatar[name]('x');
53
- return { threw: false, message: '' };
54
  } catch (err) {
55
- return { threw: true, message: String((err && err.message) || err) };
56
  }
 
 
 
 
 
 
 
 
 
 
 
57
  }
58
  """
59
 
60
- # Type check plus a real call. A method that exists but still throws the not-wired error
61
- # would pass a typeof check and fail a learner, so both halves are needed.
62
- PROBE_WIRED = """
63
- async (name) => {
64
- const isFunction = typeof window.Avatar[name] === 'function';
65
- let notWired = false;
66
- let message = '';
67
  try {
68
- await window.Avatar[name]();
69
  } catch (err) {
70
- message = String((err && err.message) || err);
71
- notWired = message.includes('not wired yet');
72
  }
73
- return { isFunction, notWired, message };
 
 
 
 
 
 
 
 
 
74
  }
75
  """
76
 
77
 
 
 
 
 
 
 
 
 
 
 
78
  @pytest.fixture(scope="session")
79
  def live_avatars(browser, gradio_apps):
80
- """Boot each transport once and capture everything the three tests compare."""
81
  captured = {}
82
  for transport in TRANSPORTS:
83
  url = gradio_apps(transport)
84
  page = browser.new_page()
 
85
  try:
86
  page.goto(url)
87
  page.wait_for_function(AVATAR_READY, timeout=BOOT_TIMEOUT_MS)
88
- captured[transport] = {
89
  "surface": page.evaluate(
90
  "() => Object.keys(window.Avatar).filter(k => k !== '__debug').sort()"
91
  ),
@@ -93,18 +243,42 @@ def live_avatars(browser, gradio_apps):
93
  "async () => Object.keys(await window.Avatar.getDebug()).sort()"
94
  ),
95
  "reported_transport": page.evaluate("() => window.Avatar.__debug.transport"),
96
- "deferred": {name: page.evaluate(CALL_DEFERRED, name) for name in DEFERRED_METHODS},
97
- "wired": {name: page.evaluate(PROBE_WIRED, name) for name in WIRED_METHODS},
98
  }
99
  page.wait_for_function(FIRST_FRAME, timeout=FIRST_FRAME_TIMEOUT_MS)
100
- captured[transport]["arm_down"] = page.evaluate(
101
  "async () => (await window.Avatar.getDebug()).armDown"
102
  )
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
103
  finally:
104
  page.close()
105
  return captured
106
 
107
 
 
 
 
 
 
 
 
 
108
  def test_transports_expose_identical_surfaces(live_avatars):
109
  inline = live_avatars["inline"]["surface"]
110
  iframe = live_avatars["iframe"]["surface"]
@@ -129,6 +303,8 @@ def test_transports_expose_identical_debug_keys(live_avatars):
129
  "__debug has drifted between transports; symmetric difference "
130
  f"(inline ^ iframe) = {sorted(set(inline) ^ set(iframe))}"
131
  )
 
 
132
 
133
 
134
  def test_arms_rest_at_sides_under_both_transports(live_avatars):
@@ -148,38 +324,129 @@ def test_arms_rest_at_sides_under_both_transports(live_avatars):
148
  )
149
 
150
 
151
- def test_deferred_methods_fail_loudly_not_silently(live_avatars):
152
- """The remaining stubs are placeholders, not accidental no-ops.
153
 
154
- Scope shrinks with each wave rather than the test being deleted: plan 01-07 removed
155
- startListening/stopListening from DEFERRED_METHODS when it implemented them, and plan
156
- 01-08 empties the tuple. Whatever is still deferred must still fail loudly.
 
157
  """
158
  for transport in TRANSPORTS:
159
- for name, outcome in live_avatars[transport]["deferred"].items():
160
- assert outcome["threw"], (
161
- f"{transport}: Avatar.{name}() returned instead of throwing; a silent "
162
- "no-op is exactly what the deferred stubs exist to prevent"
163
  )
164
- assert DEFERRED_PLAN in outcome["message"], (
165
- f"{transport}: Avatar.{name}() threw {outcome['message']!r}, which does "
166
- f"not name plan {DEFERRED_PLAN}"
167
  )
168
 
169
 
170
- def test_push_to_talk_exists_under_both_transports(live_avatars):
171
- """The payoff of putting the wiring in avatar/turn-loop.js.
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
172
 
173
- The iframe transport gained push-to-talk without gaining a line of code, and this is
174
- the assertion that proves it against a LIVE object rather than against a grep.
 
 
 
 
 
175
  """
176
  for transport in TRANSPORTS:
177
- for name, outcome in live_avatars[transport]["wired"].items():
178
- assert outcome["isFunction"], (
179
- f"{transport}: Avatar.{name} is not a function; the shared turn loop did "
180
- "not reach this transport"
181
- )
182
- assert not outcome["notWired"], (
183
- f"{transport}: Avatar.{name}() still throws the plan {WIRED_PLAN} "
184
- f"not-wired error: {outcome['message']!r}"
185
- )
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Mechanical proof that the two transports expose the same object - and the same turn.
2
 
3
  The static tests in tests/test_transport_seam.py can only prove that no transport
4
  *writes* window.Avatar. These boot the real Gradio app twice - once per
5
  AVATAR_TRANSPORT value - and compare the LIVE objects, which is the only way to catch
6
  a method that resolves in one transport and silently does not in the other.
7
 
8
+ Since wave 5 the same fixture also drives a real turn through each transport: text in,
9
+ synthesised speech out, thinking pose engaged in between, then a networkless replay and
10
+ a slower re-read. Those are the local rehearsal, at the same thresholds, of the deployed
11
+ rows plan 01-09 binds (test_text_turn, test_thinking_state, test_replay, test_slower),
12
+ run under BOTH transports so a spike reversal could never cost the turn loop.
13
+
14
+ Marked slow but NOT deployed: they run entirely locally.
15
  """
16
 
17
  from __future__ import annotations
18
 
19
  import pytest
20
 
21
+ from tests.e2e.test_stage_standalone import (
22
+ ARM_DOWN_MAX,
23
+ FIRST_FRAME_TIMEOUT_MS,
24
+ HEAD_PITCH_IDLE_MAX,
25
+ HEAD_PITCH_THINKING_MIN,
26
+ )
27
  from tests.test_transport_seam import avatar_surface
28
 
29
  pytestmark = pytest.mark.slow
 
31
  TRANSPORTS = ("inline", "iframe")
32
  AVATAR_READY = "() => !!window.Avatar && !!window.Avatar.__debug && window.Avatar.__debug.ready"
33
  BOOT_TIMEOUT_MS = 120_000
34
+ TURN_TIMEOUT_MS = 120_000
35
+
36
+ # こんにちは: the golden fixture sentence, so the viseme sequence and the slow/normal
37
+ # ratio are already pinned by the unit suite (tests/test_directive.py).
38
+ TURN_TEXT = "こんにちは"
39
+ SLOWER_SPEED = 0.75
40
+ # The realised ratio is near 1/0.75, never exactly on it (VOICEVOX re-quantises after
41
+ # dividing); 5% is the deployed band plan 01-09 uses, so the rehearsal matches it.
42
+ SLOWER_RATIO_TOLERANCE = 0.05
43
 
44
  # The pose is measured after a render, and ready fires before the first one. Same
45
  # frame-rendered signal as the standalone suite, read through the facade because the
 
51
  }
52
  """
53
 
54
+ # Type check plus a real call, for EVERY name in the surface. A method that exists but
55
+ # still throws the not-wired error would pass a typeof check and fail a learner, so both
56
+ # halves are needed. Called with no arguments: most reject for an ordinary reason (no
57
+ # text, nothing cached, no microphone), and that is fine - only the deferred-stub error
58
+ # is a failure. mount is excluded from the call because a second mount is exactly the
59
+ # remount AVTR-01 forbids; its type is still checked.
60
+ PROBE_SURFACE = """
61
+ async (names) => {
62
+ const out = {};
63
+ for (const name of names) {
64
+ const isFunction = typeof window.Avatar[name] === 'function';
65
+ let notWired = false;
66
+ let message = '';
67
+ if (isFunction && name !== 'mount') {
68
+ try {
69
+ await window.Avatar[name]();
70
+ } catch (err) {
71
+ message = String((err && err.message) || err);
72
+ notWired = message.includes('not wired yet');
73
+ }
74
+ }
75
+ out[name] = { isFunction, notWired, message };
76
+ }
77
+ return out;
78
+ }
79
+ """
80
+
81
+ # One full text turn, observed the way a learner experiences it: thinking engages at
82
+ # dispatch (before any await), the head visibly tilts while the server works, speech
83
+ # starts, the mouth opens to distinct shapes, speech ends, thinking is long gone.
84
+ # Everything is read through getDebug() per animation frame, never from the stale
85
+ # __debug snapshot. Note that getDebug() resolves to the facade's LIVE merged object,
86
+ # not a copy: every scalar is read out the moment it is sampled, because holding the
87
+ # object and reading it later reads the final state.
88
+ TURN_PROBE = """
89
+ async ({ text, timeoutMs }) => {
90
+ const events = [];
91
+ const names = ['turn-start', 'turn', 'speech-start', 'speech-end', 'latency', 'error'];
92
+ const offs = names.map((n) =>
93
+ window.Avatar.on(n, (d) => events.push({ name: n, at: performance.now(), data: d ?? null }))
94
+ );
95
+ const t0 = performance.now();
96
+ const turn = window.Avatar.dispatchTurn(text);
97
+ const immediateThinking = (await window.Avatar.getDebug()).thinking;
98
+ const immediateAt = performance.now();
99
+
100
+ let maxHeadPitch = -1;
101
+ let maxRelaxed = 0;
102
+ let thinkingSamples = 0;
103
+ let samples = 0;
104
+ let thinkingAtSpeechStart = null;
105
+ while (!events.some((e) => e.name === 'speech-start') && performance.now() - t0 < timeoutMs) {
106
+ const d = await window.Avatar.getDebug();
107
+ samples += 1;
108
+ if (d.thinking) {
109
+ thinkingSamples += 1;
110
+ maxHeadPitch = Math.max(maxHeadPitch, d.headPitch);
111
+ maxRelaxed = Math.max(maxRelaxed, d.relaxedValue);
112
+ }
113
+ await new Promise((r) => requestAnimationFrame(r));
114
+ }
115
+ thinkingAtSpeechStart = (await window.Avatar.getDebug()).thinking;
116
+
117
+ let result = null;
118
+ let error = null;
119
+ try {
120
+ result = await turn;
121
+ } catch (err) {
122
+ error = String((err && err.message) || err);
123
+ }
124
+ // speak() resolves at speech-end, while the player is still cross-fading the mouth
125
+ // shut (a 50 ms attack); give it up to 2 s of frames to settle, as the standalone
126
+ // suite does, then read the final state.
127
+ const settledBy = performance.now() + 2000;
128
+ let after = await window.Avatar.getDebug();
129
+ while (
130
+ performance.now() < settledBy &&
131
+ Object.values(after.currentVisemes).some((v) => v !== 0)
132
+ ) {
133
+ await new Promise((r) => requestAnimationFrame(r));
134
+ after = await window.Avatar.getDebug();
135
+ }
136
+ for (const off of offs) off();
137
+ return {
138
+ error,
139
+ result,
140
+ immediateThinking,
141
+ immediateMs: Math.round(immediateAt - t0),
142
+ thinkingSamples,
143
+ samples,
144
+ maxHeadPitch,
145
+ maxRelaxed,
146
+ thinkingAtSpeechStart,
147
+ events: events.map((e) => ({
148
+ name: e.name,
149
+ at: Math.round(e.at - t0),
150
+ duration: e.data && typeof e.data.duration === 'number' ? e.data.duration : null,
151
+ })),
152
+ after: {
153
+ thinking: after.thinking,
154
+ speaking: after.speaking,
155
+ headPitch: after.headPitch,
156
+ turnCount: after.turnCount,
157
+ lastSubtitle: after.lastSubtitle,
158
+ lastSpeed: after.lastSpeed,
159
+ lastTurnMs: after.lastTurnMs,
160
+ lastStageTimings: after.lastStageTimings,
161
+ visemePeaks: after.visemePeaks,
162
+ currentVisemes: after.currentVisemes,
163
+ transport: after.transport,
164
+ },
165
+ };
166
+ }
167
+ """
168
+
169
+ REPLAY_PROBE = """
170
+ async () => {
171
+ const events = [];
172
+ const off = window.Avatar.on('speech-end', (d) => events.push(d));
173
+ const before = await window.Avatar.getDebug();
174
+ let error = null;
175
  try {
176
+ await window.Avatar.replay();
 
177
  } catch (err) {
178
+ error = String((err && err.message) || err);
179
  }
180
+ const after = await window.Avatar.getDebug();
181
+ off();
182
+ return {
183
+ error,
184
+ speechEnds: events.length,
185
+ duration: events[0] ? events[0].duration : null,
186
+ turnCountBefore: before.turnCount,
187
+ turnCountAfter: after.turnCount,
188
+ replayCount: after.replayCount,
189
+ lastReplayMs: after.lastReplayMs,
190
+ };
191
  }
192
  """
193
 
194
+ SLOWER_PROBE = """
195
+ async () => {
196
+ let error = null;
197
+ let result = null;
 
 
 
198
  try {
199
+ result = await window.Avatar.requestSlower();
200
  } catch (err) {
201
+ error = String((err && err.message) || err);
 
202
  }
203
+ const after = await window.Avatar.getDebug();
204
+ return {
205
+ error,
206
+ result,
207
+ turnCount: after.turnCount,
208
+ lastSpeed: after.lastSpeed,
209
+ lastSubtitle: after.lastSubtitle,
210
+ lastStageTimings: after.lastStageTimings,
211
+ visemePeaks: after.visemePeaks,
212
+ };
213
  }
214
  """
215
 
216
 
217
+ def _wait_settled(page):
218
+ """The controls re-arm 200 ms after speech-end; a follow-on call must not race that."""
219
+ page.wait_for_function(
220
+ "async () => { const d = await window.Avatar.getDebug(); "
221
+ "return !d.thinking && !d.speaking; }",
222
+ timeout=TURN_TIMEOUT_MS,
223
+ )
224
+ page.wait_for_timeout(300)
225
+
226
+
227
  @pytest.fixture(scope="session")
228
  def live_avatars(browser, gradio_apps):
229
+ """Boot each transport once and capture everything the tests compare."""
230
  captured = {}
231
  for transport in TRANSPORTS:
232
  url = gradio_apps(transport)
233
  page = browser.new_page()
234
+ requests: list[str] = []
235
  try:
236
  page.goto(url)
237
  page.wait_for_function(AVATAR_READY, timeout=BOOT_TIMEOUT_MS)
238
+ record = {
239
  "surface": page.evaluate(
240
  "() => Object.keys(window.Avatar).filter(k => k !== '__debug').sort()"
241
  ),
 
243
  "async () => Object.keys(await window.Avatar.getDebug()).sort()"
244
  ),
245
  "reported_transport": page.evaluate("() => window.Avatar.__debug.transport"),
246
+ "probe": page.evaluate(PROBE_SURFACE, avatar_surface()),
 
247
  }
248
  page.wait_for_function(FIRST_FRAME, timeout=FIRST_FRAME_TIMEOUT_MS)
249
+ record["arm_down"] = page.evaluate(
250
  "async () => (await window.Avatar.getDebug()).armDown"
251
  )
252
+ record["idle_head_pitch"] = page.evaluate(
253
+ "async () => (await window.Avatar.getDebug()).headPitch"
254
+ )
255
+
256
+ # The turn. Runs even where synthesis is unavailable; the turn tests then
257
+ # skip on the recorded error rather than the whole parity suite failing.
258
+ record["turn"] = page.evaluate(
259
+ TURN_PROBE, {"text": TURN_TEXT, "timeoutMs": TURN_TIMEOUT_MS}
260
+ )
261
+ if record["turn"]["error"] is None:
262
+ _wait_settled(page)
263
+ page.on("request", lambda r, sink=requests.append: sink(r.url))
264
+ record["replay"] = page.evaluate(REPLAY_PROBE)
265
+ record["replay_requests"] = list(requests)
266
+ _wait_settled(page)
267
+ record["slower"] = page.evaluate(SLOWER_PROBE)
268
+ captured[transport] = record
269
  finally:
270
  page.close()
271
  return captured
272
 
273
 
274
+ def _turn_or_skip(live_avatars, transport):
275
+ turn = live_avatars[transport]["turn"]
276
+ if turn["error"] is not None:
277
+ pytest.importorskip("voicevox_core")
278
+ pytest.fail(f"{transport}: dispatchTurn rejected: {turn['error']}")
279
+ return turn
280
+
281
+
282
  def test_transports_expose_identical_surfaces(live_avatars):
283
  inline = live_avatars["inline"]["surface"]
284
  iframe = live_avatars["iframe"]["surface"]
 
303
  "__debug has drifted between transports; symmetric difference "
304
  f"(inline ^ iframe) = {sorted(set(inline) ^ set(iframe))}"
305
  )
306
+ for key in ("lastTurnMs", "lastStageTimings", "lastSubtitle", "replayCount", "turnCount"):
307
+ assert key in inline, f"__debug.{key} is missing from the live object"
308
 
309
 
310
  def test_arms_rest_at_sides_under_both_transports(live_avatars):
 
324
  )
325
 
326
 
327
+ def test_turn_surface_is_live_under_both_transports(live_avatars):
328
+ """Every name in AVATAR_SURFACE is a function and none is a deferred stub - on BOTH.
329
 
330
+ This is the mechanical proof that a spike failure would not have cost VOIC-02/03/04/05:
331
+ the iframe transport gained the whole turn loop without gaining a line of code, and
332
+ this asserts it against a LIVE object rather than against a grep. It replaces the
333
+ deferred-stub guard that shrank wave by wave and is empty now.
334
  """
335
  for transport in TRANSPORTS:
336
+ for name, outcome in live_avatars[transport]["probe"].items():
337
+ assert outcome["isFunction"], (
338
+ f"{transport}: Avatar.{name} is not a function; the shared turn loop did "
339
+ "not reach this transport"
340
  )
341
+ assert not outcome["notWired"], (
342
+ f"{transport}: Avatar.{name}() still throws a not-wired error: "
343
+ f"{outcome['message']!r}"
344
  )
345
 
346
 
347
+ def test_text_turn_speaks_under_both_transports(live_avatars):
348
+ """VOIC-04 rehearsal: text in, speech-start then speech-end, the mouth actually moved."""
349
+ for transport in TRANSPORTS:
350
+ turn = _turn_or_skip(live_avatars, transport)
351
+ names = [e["name"] for e in turn["events"]]
352
+ print(f"[{transport}] turn events: {turn['events']}")
353
+ print(f"[{transport}] after: {turn['after']}")
354
+
355
+ assert "turn-start" in names and "turn" in names, names
356
+ assert "speech-start" in names and "speech-end" in names, (
357
+ f"{transport}: the turn never produced speech: {names}"
358
+ )
359
+ assert names.index("speech-start") < names.index("speech-end")
360
+ assert "error" not in names, [e for e in turn["events"] if e["name"] == "error"]
361
+
362
+ after = turn["after"]
363
+ assert after["transport"] == transport
364
+ assert after["turnCount"] == 1
365
+ assert after["lastSubtitle"] == TURN_TEXT
366
+ assert after["lastSpeed"] == 1.0
367
+ assert not after["speaking"]
368
+ # The number that matters, and the server breakdown that travelled with it.
369
+ assert isinstance(after["lastTurnMs"], int | float) and after["lastTurnMs"] > 0
370
+ assert set(after["lastStageTimings"]) >= {
371
+ "audio_query_ms",
372
+ "synthesis_ms",
373
+ "timeline_ms",
374
+ "encode_ms",
375
+ "server_total_ms",
376
+ }
377
+ assert after["lastStageTimings"]["synthesis_ms"] > 0
378
+ # こんにちは drives o, i, i, a: three distinct shapes, not one flap.
379
+ peaks = after["visemePeaks"]
380
+ opened = [n for n in ("aa", "ih", "oh") if peaks.get(n, 0) > 0.5]
381
+ assert len(opened) == 3, f"{transport}: only {opened} opened; peaks {peaks}"
382
+ assert all(v == 0 for v in after["currentVisemes"].values()), (
383
+ f"{transport}: the mouth did not shut after speech-end: {after['currentVisemes']}"
384
+ )
385
 
386
+
387
+ def test_thinking_state_under_both_transports(live_avatars):
388
+ """VOIC-05 rehearsal: thinking engages at dispatch and clears at speech-start.
389
+
390
+ Asserted as numbers at this layer too: `thinking` is true on the first getDebug()
391
+ after dispatch, the head is measurably pitched down while it is true, and it is
392
+ false by the time speech starts and stays false afterwards.
393
  """
394
  for transport in TRANSPORTS:
395
+ turn = _turn_or_skip(live_avatars, transport)
396
+ print(
397
+ f"[{transport}] thinking: immediate={turn['immediateThinking']} "
398
+ f"({turn['immediateMs']} ms after dispatch), {turn['thinkingSamples']}/"
399
+ f"{turn['samples']} frames thinking, maxHeadPitch={turn['maxHeadPitch']:.3f}, "
400
+ f"maxRelaxed={turn['maxRelaxed']}, idleHeadPitch="
401
+ f"{live_avatars[transport]['idle_head_pitch']:.3f}"
402
+ )
403
+ assert turn["immediateThinking"] is True, (
404
+ f"{transport}: thinking was not true on the first getDebug() after dispatch "
405
+ f"({turn['immediateMs']} ms later)"
406
+ )
407
+ assert turn["thinkingSamples"] > 0
408
+ assert turn["maxHeadPitch"] > HEAD_PITCH_THINKING_MIN, (
409
+ f"{transport}: the head never pitched down while thinking "
410
+ f"(max {turn['maxHeadPitch']:.3f}); the pose was requested but not rendered"
411
+ )
412
+ assert turn["maxRelaxed"] > 0
413
+ assert turn["thinkingAtSpeechStart"] is False, (
414
+ f"{transport}: thinking was still true when speech started"
415
+ )
416
+ assert turn["after"]["thinking"] is False
417
+ assert abs(turn["after"]["headPitch"]) < HEAD_PITCH_IDLE_MAX
418
+ assert abs(live_avatars[transport]["idle_head_pitch"]) < HEAD_PITCH_IDLE_MAX
419
+
420
+
421
+ def test_replay_is_networkless_under_both_transports(live_avatars):
422
+ """VOIC-03 rehearsal: replay re-plays the cached buffer with zero requests."""
423
+ for transport in TRANSPORTS:
424
+ _turn_or_skip(live_avatars, transport)
425
+ replay = live_avatars[transport]["replay"]
426
+ requests = live_avatars[transport]["replay_requests"]
427
+ print(f"[{transport}] replay: {replay}; requests during replay: {requests}")
428
+ assert replay["error"] is None, f"{transport}: replay rejected: {replay['error']}"
429
+ assert replay["speechEnds"] == 1
430
+ assert replay["replayCount"] == 1
431
+ assert replay["turnCountAfter"] == replay["turnCountBefore"], "a replay is not a turn"
432
+ assert requests == [], f"{transport}: replay made network requests: {requests}"
433
+ assert replay["lastReplayMs"] is not None and replay["lastReplayMs"] < 1000
434
+
435
+
436
+ def test_slower_resynthesises_under_both_transports(live_avatars):
437
+ """VOIC-03 rehearsal: the slow re-read is longer audio from a real re-synthesis."""
438
+ for transport in TRANSPORTS:
439
+ turn = _turn_or_skip(live_avatars, transport)
440
+ slower = live_avatars[transport]["slower"]
441
+ assert slower["error"] is None, f"{transport}: requestSlower rejected: {slower['error']}"
442
+ normal = turn["result"]["duration"]
443
+ slow = slower["result"]["duration"]
444
+ ratio = slow / normal
445
+ print(f"[{transport}] slower: normal {normal:.3f}s, slow {slow:.3f}s, ratio {ratio:.4f}")
446
+
447
+ assert slower["turnCount"] == 2, "slower is a turn - it must round-trip to the server"
448
+ assert slower["lastSpeed"] == SLOWER_SPEED
449
+ assert slower["lastSubtitle"] == TURN_TEXT
450
+ assert abs(ratio - 1 / SLOWER_SPEED) < SLOWER_RATIO_TOLERANCE * (1 / SLOWER_SPEED), ratio
451
+ assert slower["lastStageTimings"]["synthesis_ms"] > 0, "not re-synthesised server-side"
452
+ assert any(v > 0.4 for v in slower["visemePeaks"].values())
tests/e2e/test_stage_standalone.py CHANGED
@@ -24,6 +24,13 @@ STAGE_READY = "() => !!window.__stageDebug && window.__stageDebug.ready === true
24
  # idle sway. Shared verbatim with the deployed suite.
25
  ARM_DOWN_MAX = -0.7
26
 
 
 
 
 
 
 
 
27
  # `ready` is announced before the first frame renders, and the first frame compiles every
28
  # MToon shader - measured at over 1.5 s on the headless fleet - so the debug snapshot can
29
  # still hold its boot-time zeros well after ready. breathValue is written on every tick and
@@ -210,3 +217,42 @@ def test_console_has_no_multiple_three_warning(page, static_server):
210
 
211
  offenders = [m for m in messages if "Multiple instances of Three.js" in m]
212
  assert not offenders, f"three.js reported duplicate instances: {offenders}"
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
24
  # idle sway. Shared verbatim with the deployed suite.
25
  ARM_DOWN_MAX = -0.7
26
 
27
+ # headPitch is the downward component of the head's world-space forward direction:
28
+ # ~sin(THINK_TILT) = 0.08 while thinking, ~0 at rest, plus or minus the 0.012 rad breath.
29
+ # 0.05 / 0.03 leave the breath a comfortable margin on both sides. Shared with the parity
30
+ # suite so the three layers assert the same numbers.
31
+ HEAD_PITCH_THINKING_MIN = 0.05
32
+ HEAD_PITCH_IDLE_MAX = 0.03
33
+
34
  # `ready` is announced before the first frame renders, and the first frame compiles every
35
  # MToon shader - measured at over 1.5 s on the headless fleet - so the debug snapshot can
36
  # still hold its boot-time zeros well after ready. breathValue is written on every tick and
 
217
 
218
  offenders = [m for m in messages if "Multiple instances of Three.js" in m]
219
  assert not offenders, f"three.js reported duplicate instances: {offenders}"
220
+
221
+
222
+ def test_thinking_pose_is_visible(page, static_server):
223
+ """VOIC-05's thinking state, as a rendered number rather than a flag.
224
+
225
+ The standalone harness has no server, so it cannot dispatch a turn - but the pose
226
+ the turn engages is a stage property, and the harness answers the same postMessage
227
+ protocol the iframe transport speaks, from its own window. setThinking(true) must
228
+ pitch the rendered head down and raise the 'relaxed' expression; setThinking(false)
229
+ must return both to rest. A flag that says "thinking" with no visible change is the
230
+ T-pose lesson again.
231
+ """
232
+ _open_stage(page, static_server)
233
+ page.wait_for_function(FIRST_FRAME, timeout=FIRST_FRAME_TIMEOUT_MS)
234
+ rest = page.evaluate("() => window.__stageDebug.headPitch")
235
+ assert abs(rest) < HEAD_PITCH_IDLE_MAX, f"head pitched {rest:.3f} at rest"
236
+
237
+ page.evaluate(
238
+ "() => window.postMessage({ type: 'avatar:setThinking', value: true, id: 'think' }, '*')"
239
+ )
240
+ page.wait_for_function(
241
+ f"() => window.__stageDebug.headPitch > {HEAD_PITCH_THINKING_MIN}", timeout=5_000
242
+ )
243
+ thinking = page.evaluate(
244
+ "() => ({ headPitch: window.__stageDebug.headPitch, "
245
+ "relaxed: window.__stageDebug.relaxedValue, flag: window.__stageDebug.thinking })"
246
+ )
247
+ print(f"[standalone] thinking pose: {thinking}")
248
+ assert thinking["flag"] is True
249
+ assert thinking["relaxed"] > 0
250
+
251
+ page.evaluate(
252
+ "() => window.postMessage({ type: 'avatar:setThinking', value: false, id: 'idle' }, '*')"
253
+ )
254
+ page.wait_for_function(
255
+ f"() => Math.abs(window.__stageDebug.headPitch) < {HEAD_PITCH_IDLE_MAX}", timeout=5_000
256
+ )
257
+ assert page.evaluate("() => window.__stageDebug.relaxedValue") == 0
258
+ assert page.evaluate("() => window.__stageDebug.thinking") is False
tests/test_transport_seam.py CHANGED
@@ -24,6 +24,11 @@ TURN_SURFACE = ["startListening", "stopListening", "dispatchTurn", "requestSlowe
24
  # only rendering and audio playback live inside the iframe - so these modules must hang
25
  # off the shared turn loop, never off a transport.
26
  AUDIO_IN = ["mic.js", "asr.js"]
 
 
 
 
 
27
 
28
 
29
  def src(name: str) -> str:
@@ -269,3 +274,68 @@ def test_transports_gained_push_to_talk_without_gaining_code(name):
269
  assert token not in s, (
270
  f"{name} mentions {token!r}; push-to-talk must live only in avatar/turn-loop.js"
271
  )
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
24
  # only rendering and audio playback live inside the iframe - so these modules must hang
25
  # off the shared turn loop, never off a transport.
26
  AUDIO_IN = ["mic.js", "asr.js"]
27
+ # Wave 5. The host glue binds the page's controls to window.Avatar. It is loaded by the
28
+ # shared boot template in avatar_component.py, so both transports get the same controls;
29
+ # it must import no avatar module, be imported by none, and implement no turn behaviour.
30
+ HOST = "host.js"
31
+ COMPONENT = AVATAR.parent / "src" / "japanese_avatar" / "ui" / "avatar_component.py"
32
 
33
 
34
  def src(name: str) -> str:
 
274
  assert token not in s, (
275
  f"{name} mentions {token!r}; push-to-talk must live only in avatar/turn-loop.js"
276
  )
277
+
278
+
279
+ @pytest.mark.parametrize("member", ["dispatchTurn", "requestSlower"])
280
+ def test_turn_dispatch_is_implemented_not_deferred(member):
281
+ """The wave-5 methods are real, and nothing in the file is a deferred stub any more."""
282
+ s = src("turn-loop.js")
283
+ assert "notWiredYet" not in s, "a deferred stub survives in avatar/turn-loop.js"
284
+ assert f"{member}(" in s, f"{member} must be implemented in avatar/turn-loop.js"
285
+
286
+
287
+ def test_thinking_engages_before_the_first_await():
288
+ """VOIC-05's cheapest latency mitigation: the pose switches at dispatch, not at response.
289
+
290
+ Verified by line number inside dispatchTurn: setThinking(true) precedes the first
291
+ await, which is the server call.
292
+ """
293
+ lines = src("turn-loop.js").splitlines()
294
+ start = next(i for i, line in enumerate(lines) if "async function dispatchTurn(" in line)
295
+ body = lines[start:]
296
+ think_at = next(i for i, line in enumerate(body) if "setThinking(true)" in line)
297
+ await_at = next(i for i, line in enumerate(body) if "await " in line)
298
+ assert think_at < await_at, (
299
+ f"setThinking(true) is on dispatchTurn line {think_at} but the first await is on "
300
+ f"line {await_at}; the thinking pose must engage before the server round trip"
301
+ )
302
+ assert "await bridge." in body[await_at], body[await_at]
303
+
304
+
305
+ def test_replay_is_networkless_by_construction():
306
+ """replay() re-emits the cached buffer. No fetch, no dynamic import, anywhere in the loop."""
307
+ s = src("turn-loop.js")
308
+ assert "fetch(" not in s
309
+ assert "import(" not in s
310
+ assert "XMLHttpRequest" not in s
311
+
312
+
313
+ def test_turn_loop_publishes_the_latency_numbers():
314
+ s = src("turn-loop.js")
315
+ for key in ("lastTurnMs", "lastStageTimings", "lastSubtitle", "replayCount", "turnCount"):
316
+ assert f"{key}:" in s, f"__debug.{key} is not initialised in avatar/turn-loop.js"
317
+ for mark in ("turn:dispatch", "turn:response", "turn:speech-start"):
318
+ assert f"performance.mark('{mark}')" in s, f"performance.mark({mark!r}) missing"
319
+ assert "0.75" in s, "the slower speed must be the VOICEVOX speedScale 0.75"
320
+ assert "greeting" in s
321
+
322
+
323
+ def test_host_glue_is_neither_a_transport_nor_the_turn_loop():
324
+ """host.js is page glue: loaded from the shared boot template, importing nothing here."""
325
+ host = src(HOST)
326
+ assert not re.search(r"^\s*import\s", host, re.M), (
327
+ "host.js must not import any avatar module; it is handed the facade"
328
+ )
329
+ for member in TURN_SURFACE:
330
+ assert f"function {member}" not in host and f"async {member}(" not in host, (
331
+ f"host.js implements {member}; turn behaviour belongs in avatar/turn-loop.js"
332
+ )
333
+ for member in ["dispatchTurn", "requestSlower", "replay", "startListening", "stopListening"]:
334
+ assert f".{member}(" in host, f"host.js never calls Avatar.{member}()"
335
+ for name in [*TRANSPORTS, *SHARED, *CORE, *AUDIO_IN]:
336
+ assert HOST not in src(name), f"{name} references {HOST}; only the boot template may"
337
+ component = COMPONENT.read_text(encoding="utf-8")
338
+ assert component.count(f"import('/gradio_api/file=avatar/{HOST}')") == 1, (
339
+ "the shared boot template must load host.js exactly once, for both transports"
340
+ )
341
+ assert "bindHost(avatar, document)" in component