File size: 15,385 Bytes
371c5a6
 
 
 
 
 
 
c6dc487
371c5a6
 
 
c6dc487
 
6c967d2
 
 
 
 
 
 
 
371c5a6
 
 
 
6c967d2
371c5a6
6c967d2
371c5a6
 
 
 
 
 
 
6c967d2
 
 
371c5a6
 
 
 
6c967d2
 
 
 
 
371c5a6
 
 
 
 
 
 
 
 
 
 
 
 
 
6c967d2
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
371c5a6
 
 
 
 
 
 
 
 
 
 
 
 
c6dc487
371c5a6
 
 
 
 
 
 
 
 
 
 
 
 
 
 
6c967d2
 
 
 
 
 
371c5a6
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
6c967d2
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
"""AVTR-02 and VOIC-03: the viseme timeline builder.

This is the densest test file in Phase 1 because ``visemes.py`` is the one function whose
failure modes are all silent. A wrong rounding mode, a lowercase-only vowel table or a
float-accumulating loop each produce a timeline that looks entirely plausible in a debugger and
visibly wrong on the avatar's face.

Every fixture here was captured from a real VOICEVOX CORE 0.17.0 synthesis by
``tests/fixtures/make_synth_fixtures.py``; every duration compared against is read from the WAV
header, never summed from the query.

Nothing in this file imports the VOICEVOX wheel - the builder is a pure function over plain
dicts, so the quick loop runs on a machine that has never installed it.

Regenerating the fixtures (a maintainer action, not CI)::

    uv run --extra voice python tests/fixtures/make_synth_fixtures.py
    uv run python tests/fixtures/make_golden_timeline.py

Run them in that order: the golden timeline is built from ``audio_query_short.json``, so it is
downstream of the synthesis fixtures.
"""

from __future__ import annotations

import json
import math
from pathlib import Path

import pytest

from japanese_avatar.voice.visemes import (
    DEVOICED,
    FRAMERATE,
    VOWEL_TO_VISEME,
    build_timeline,
    frames_for,
    timeline_to_dicts,
    to_frame,
    viseme_for,
)

FIXTURES = Path(__file__).parent / "fixtures"

#: One VOICEVOX frame. The tolerance every duration assertion in this file is written against.
ONE_FRAME = 1 / FRAMERATE

# frames -> to_frame(frames / FRAMERATE). Round-half-up would give the third column.
# Verified to round-trip exactly through f / 93.75 * 93.75 on CPython.
BANKERS_VECTORS = [
    (0.5, 0, 1),
    (1.5, 2, 2),
    (2.5, 2, 3),
    (4.5, 4, 5),
    (5.5, 6, 6),
    (10.5, 10, 11),
    (20.5, 20, 21),
    (100.5, 100, 101),
]


def _query(case: str) -> dict:
    return json.loads((FIXTURES / f"audio_query_{case}.json").read_text(encoding="utf-8"))


def _meta() -> dict:
    return json.loads((FIXTURES / "synth_meta.json").read_text(encoding="utf-8"))


def _mora(text: str, vowel: str, vowel_length: float, consonant=None, consonant_length=None):
    return {
        "text": text,
        "vowel": vowel,
        "vowel_length": vowel_length,
        "consonant": consonant,
        "consonant_length": consonant_length,
        "pitch": 5.5,
    }


def test_vowel_mapping():
    """AVTR-02: every VOICEVOX vowel symbol, including the devoiced uppercase ones."""
    assert VOWEL_TO_VISEME["a"] == "aa"
    assert VOWEL_TO_VISEME["i"] == "ih"
    assert VOWEL_TO_VISEME["u"] == "ou"
    assert VOWEL_TO_VISEME["e"] == "ee"
    assert VOWEL_TO_VISEME["o"] == "oh"

    # Japanese devoices /i/ and /u/ between voiceless consonants constantly - です is
    # "d e s U", した is "sh I t a". VOICEVOX emits those as UPPERCASE. A lowercase-only
    # table freezes the mouth on every polite form.
    for v in "AIUEO":
        assert VOWEL_TO_VISEME[v] == VOWEL_TO_VISEME[v.lower()], v
    assert frozenset("AIUEO") == DEVOICED

    assert VOWEL_TO_VISEME["N"] == "closed"  # ん
    assert VOWEL_TO_VISEME["cl"] == "closed"  # っ
    assert VOWEL_TO_VISEME["pau"] == "closed"  # silence

    # Exactly 13: an extra key is an invented symbol, a missing key is a frozen mouth.
    assert len(VOWEL_TO_VISEME) == 13, sorted(VOWEL_TO_VISEME)

    # Devoiced vowels still open the mouth, just less.
    assert viseme_for("u") == ("ou", 1.0)
    assert viseme_for("U") == ("ou", 0.5)
    assert viseme_for("I") == ("ih", 0.5)
    assert viseme_for("N") == ("closed", 0.0)
    assert viseme_for("pau") == ("closed", 0.0)

    # Same claim end-to-end: a one-mora timeline carries the reduced weight through.
    voiced = build_timeline(_one_mora_query(_mora("ス", "u", 0.1, "s", 0.05)))
    devoiced = build_timeline(_one_mora_query(_mora("ス", "U", 0.1, "s", 0.05)))
    assert [(e.viseme, e.weight) for e in voiced if e.weight] == [("ou", 1.0)]
    assert [(e.viseme, e.weight) for e in devoiced if e.weight] == [("ou", 0.5)]

    # A new VOICEVOX symbol must fail loudly rather than animate wrongly.
    with pytest.raises(KeyError):
        viseme_for("x")
    with pytest.raises(KeyError):
        VOWEL_TO_VISEME["q"]


@pytest.mark.parametrize(("frames", "expected", "half_up"), BANKERS_VECTORS)
def test_frame_quantisation_bankers(frames, expected, half_up):
    """AVTR-02: round-half-to-even, matching np.round inside VOICEVOX itself.

    The VOICEVOX source carries the warning 「NOTE: `round` は偶数丸め。移植時に取扱い注意。」
    Python's built-in round() is banker's, so a Python port matches for free; JavaScript's
    Math.round() is round-half-up and would disagree on every exact-half boundary. That is an
    independent, concrete reason this arithmetic lives in Python.
    """
    assert FRAMERATE == 93.75
    assert FRAMERATE == 24000 / 256

    seconds = frames / FRAMERATE
    # Fail loudly if a future Python stops round-tripping, rather than passing by luck.
    assert seconds * FRAMERATE == frames, (frames, seconds * FRAMERATE)

    assert to_frame(seconds) == expected

    # Provably NOT round-half-up: on the exact-half boundaries the two modes disagree.
    if expected != half_up:
        assert to_frame(seconds) != math.floor(seconds * FRAMERATE + 0.5)
        assert math.floor(seconds * FRAMERATE + 0.5) == half_up


def _one_mora_query(mora: dict, speed: float = 1.0, pre: float = 0.0, post: float = 0.0) -> dict:
    return {
        "accent_phrases": [{"moras": [mora], "accent": 1, "pause_mora": None}],
        "speedScale": speed,
        "prePhonemeLength": pre,
        "postPhonemeLength": post,
    }


def test_pause_mora_ordering():
    """AVTR-02: a phrase's pause_mora follows its moras. Inverting it shifts the whole phrase."""
    query = {
        "accent_phrases": [
            {
                "moras": [_mora("ア", "a", 0.10), _mora("キ", "i", 0.10, "k", 0.05)],
                "accent": 1,
                "pause_mora": _mora("、", "pau", 0.30),
            },
            {
                "moras": [_mora("オ", "o", 0.10)],
                "accent": 1,
                "pause_mora": None,
            },
        ],
        "speedScale": 1.0,
        "prePhonemeLength": 0.0,
        "postPhonemeLength": 0.0,
    }
    events = build_timeline(query)

    # pre/post silences are zero-length here but still emitted, so the shape is
    # [pre] a, k, i, pause, o, [post].
    assert [e.viseme for e in events] == [
        "closed",  # prePhonemeLength (0 frames)
        "aa",  # ア
        "closed",  # k
        "ih",  # キ
        "closed",  # the pause_mora - AFTER its phrase, not before
        "oh",  # オ
        "closed",  # postPhonemeLength (0 frames)
    ]

    pause = events[4]
    assert pause.viseme == "closed"
    assert pause.weight == 0.0
    assert pause.dur == frames_for(0.30) / FRAMERATE

    # The pause sits between キ and オ in time, which is the property the ordering exists for.
    # Compared in frames, not seconds: `t` comes from an exact integer frame accumulator while
    # `t + dur` is a float sum, so the two differ in the last ULP. That gap is precisely why the
    # builder accumulates integers - asserting on the float sum would be asserting on the bug.
    frames = [(round(e.t * FRAMERATE), round(e.dur * FRAMERATE)) for e in events]
    assert frames[3][0] + frames[3][1] == frames[4][0]
    assert frames[4][0] + frames[4][1] == frames[5][0]
    assert all(a[0] + a[1] == b[0] for a, b in zip(frames, frames[1:], strict=False))

    # A pause_mora has no consonant, so it contributes exactly one event.
    assert sum(1 for e in events if e.viseme == "closed" and e.dur > 0) == 2  # k and the pause


def test_pre_post_silence_scaled():
    """AVTR-02: the pre/post silences exist, bracket the utterance, and ARE scaled by speed.

    VOICEVOX inserts them before the speed step, so they are divided like any other phoneme.
    Leaving them unscaled is Pitfall 5's symptom - normal speed syncs, slow speed drifts.
    """
    for case, speed in (("long", 1.0), ("slow", 0.75)):
        query = _query(case)
        assert query["speedScale"] == speed
        events = build_timeline(query)

        first, last = events[0], events[-1]
        assert first.viseme == "closed" and first.weight == 0.0
        assert last.viseme == "closed" and last.weight == 0.0
        assert first.t == 0.0

        assert first.dur == frames_for(query["prePhonemeLength"], speed) / FRAMERATE, case
        assert last.dur == frames_for(query["postPhonemeLength"], speed) / FRAMERATE, case

    # The two fixtures share a prePhonemeLength, so the silences are directly comparable:
    # 0.1 s is 9 frames at speed 1.0 and 12 frames at 0.75. Unscaled silences would be equal.
    normal = build_timeline(_query("long"))
    slow = build_timeline(_query("slow"))
    assert _query("long")["prePhonemeLength"] == _query("slow")["prePhonemeLength"]
    assert round(normal[0].dur * FRAMERATE) == 9
    assert round(slow[0].dur * FRAMERATE) == 12
    assert slow[0].dur > normal[0].dur
    assert slow[-1].dur > normal[-1].dur

    # Same claim on a synthetic query, free of any fixture-specific coincidence.
    mora = _mora("ア", "a", 0.2)
    one = build_timeline(_one_mora_query(mora, speed=1.0, pre=0.1, post=0.1))
    half = build_timeline(_one_mora_query(mora, speed=0.5, pre=0.1, post=0.1))
    assert round(one[0].dur * FRAMERATE) == 9
    assert round(half[0].dur * FRAMERATE) == 18


def test_no_drift_long_utterance():
    """AVTR-02: on a 20+ mora sentence the timeline ends where the audio ends, within a frame."""
    query = _query("long")
    meta = _meta()["cases"]["long"]
    true_duration = meta["duration_seconds"]

    events = build_timeline(query)
    assert len(events) >= 30, f"the long fixture yields only {len(events)} events; too short"

    total = sum(e.dur for e in events)
    assert abs(total - true_duration) <= ONE_FRAME, (
        f"summed timeline {total} vs true WAV duration {true_duration}: "
        f"{abs(total - true_duration) * 1000:.3f} ms, over the {ONE_FRAME * 1000:.3f} ms budget"
    )

    end = events[-1].t + events[-1].dur
    assert abs(end - true_duration) <= ONE_FRAME, (
        f"accumulated timeline ends at {end}, audio ends at {true_duration}: "
        f"{abs(end - true_duration) * 1000:.3f} ms of accumulated drift"
    )

    # Negative control: accumulate the raw floats instead of the quantised frames. If this does
    # NOT differ by more than half a frame, the fixture is too short to be catching anything.
    speed = query["speedScale"]
    lengths = [query["prePhonemeLength"], query["postPhonemeLength"]]
    for phrase in query["accent_phrases"]:
        for m in [*phrase["moras"], *([phrase["pause_mora"]] if phrase["pause_mora"] else [])]:
            lengths += [m["vowel_length"]] + ([m["consonant_length"]] if m["consonant"] else [])
    drifted = sum(x / speed for x in lengths)

    assert abs(drifted - total) > ONE_FRAME / 2, (
        f"the float-accumulating variant differs by only "
        f"{abs(drifted - total) * FRAMERATE:.4f} frames on this fixture, so the fixture is not "
        f"stressing the accumulator. Replace it with a longer sentence."
    )


def test_speed_scale():
    """VOIC-03: a 0.75x re-read matches the real slow audio and keeps the same mouth shapes."""
    meta = _meta()["cases"]
    normal = build_timeline(_query("long"))
    slow = build_timeline(_query("slow"))

    total_normal = sum(e.dur for e in normal)
    total_slow = sum(e.dur for e in slow)

    # The assertion that actually matters for lip-sync: each timeline matches its own WAV.
    assert abs(total_normal - meta["long"]["duration_seconds"]) <= ONE_FRAME
    assert abs(total_slow - meta["slow"]["duration_seconds"]) <= ONE_FRAME

    # DELIBERATE DEVIATION from 01-06-PLAN.md, which specified |total_slow - total_normal/0.75|
    # <= 2 frames. That is false and was measured to be false: VOICEVOX re-quantises every
    # phoneme AFTER dividing the frame count by speedScale, so the realised ratio lands near
    # 1/0.75 rather than on it. Measured 1.341085 vs 1.333333 - a 4-frame gap on this sentence,
    # reproduced exactly by both the builder and the engine. Pinning the measured ratio is the
    # honest test; pinning the arithmetic ideal would fail against real audio.
    # See docs/VOICEVOX-SETUP.md, "Frame quantisation".
    ratio = total_slow / total_normal
    assert ratio == pytest.approx(1.341085, abs=1e-6), ratio
    assert ratio == pytest.approx(1 / 0.75, rel=0.02), (
        f"speedScale=0.75 did not lengthen the utterance by roughly 1/0.75x: {ratio}"
    )

    # A slower re-read must not change WHICH mouth shapes appear, only how long they last.
    assert [e.viseme for e in slow] == [e.viseme for e in normal]
    assert [e.weight for e in slow] == [e.weight for e in normal]
    assert all(s.dur >= n.dur for s, n in zip(slow, normal, strict=True))


def test_golden_timeline():
    """VOIC-01: byte-stable timeline for a fixed sentence, so a core bump cannot go unnoticed."""
    golden = json.loads((FIXTURES / "golden_timeline.json").read_text(encoding="utf-8"))
    events = build_timeline(_query("short"))

    REGENERATE = (
        "The timeline for こんにちは changed. This is NOT a test to update casually: it means "
        "voicevox_core produced different mora timings than the committed fixtures, so every "
        "lip-sync in the app has shifted. Confirm the engine/model versions in "
        "tests/fixtures/synth_meta.json, then regenerate and REVIEW the diff:\n"
        "  uv run --extra voice python tests/fixtures/make_synth_fixtures.py\n"
        "  uv run python tests/fixtures/make_golden_timeline.py"
    )

    assert len(events) == len(golden), f"{len(events)} events vs {len(golden)} golden\n{REGENERATE}"
    for i, (event, want) in enumerate(zip(events, golden, strict=True)):
        # Exact float equality is correct here: every value is a whole frame count divided by
        # FRAMERATE, so any difference at all is a real timing change.
        assert event.viseme == want["viseme"], f"event {i}\n{REGENERATE}"
        assert event.weight == want["weight"], f"event {i}\n{REGENERATE}"
        assert event.t == want["t"], f"event {i}: t {event.t} vs {want['t']}\n{REGENERATE}"
        assert event.dur == want["dur"], (
            f"event {i}: dur {event.dur} vs {want['dur']}\n{REGENERATE}"
        )

    # こんにちは = k o N n i ch i w a, bracketed by the pre/post silences.
    assert [e.viseme for e in events] == [
        "closed",  # prePhonemeLength
        "closed",  # k
        "oh",  # コ
        "closed",  # ン
        "closed",  # n
        "ih",  # ニ
        "closed",  # ch
        "ih",  # チ
        "closed",  # w
        "aa",  # ワ
        "closed",  # postPhonemeLength
    ]

    # The transport form rounds t and dur to 6 dp. Sub-microsecond, so it cannot move a boundary
    # anywhere near the 10.7 ms frame; this pins that claim rather than assuming it.
    for event, compact in zip(events, timeline_to_dicts(events), strict=True):
        assert abs(compact["t"] - event.t) <= 1e-6
        assert abs(compact["dur"] - event.dur) <= 1e-6
        assert compact["viseme"] == event.viseme
        assert compact["weight"] == event.weight