"""AVTR-02 and VOIC-03: the viseme timeline builder. This is the densest test file in Phase 1 because ``visemes.py`` is the one function whose failure modes are all silent. A wrong rounding mode, a lowercase-only vowel table or a float-accumulating loop each produce a timeline that looks entirely plausible in a debugger and visibly wrong on the avatar's face. Every fixture here was captured from a real VOICEVOX CORE 0.17.0 synthesis by ``tests/fixtures/make_synth_fixtures.py``; every duration compared against is read from the WAV header, never summed from the query. Nothing in this file imports the VOICEVOX wheel - the builder is a pure function over plain dicts, so the quick loop runs on a machine that has never installed it. Regenerating the fixtures (a maintainer action, not CI):: uv run --extra voice python tests/fixtures/make_synth_fixtures.py uv run python tests/fixtures/make_golden_timeline.py Run them in that order: the golden timeline is built from ``audio_query_short.json``, so it is downstream of the synthesis fixtures. """ from __future__ import annotations import json import math from pathlib import Path import pytest from japanese_avatar.voice.visemes import ( DEVOICED, FRAMERATE, VOWEL_TO_VISEME, build_timeline, frames_for, timeline_to_dicts, to_frame, viseme_for, ) FIXTURES = Path(__file__).parent / "fixtures" #: One VOICEVOX frame. The tolerance every duration assertion in this file is written against. ONE_FRAME = 1 / FRAMERATE # frames -> to_frame(frames / FRAMERATE). Round-half-up would give the third column. # Verified to round-trip exactly through f / 93.75 * 93.75 on CPython. BANKERS_VECTORS = [ (0.5, 0, 1), (1.5, 2, 2), (2.5, 2, 3), (4.5, 4, 5), (5.5, 6, 6), (10.5, 10, 11), (20.5, 20, 21), (100.5, 100, 101), ] def _query(case: str) -> dict: return json.loads((FIXTURES / f"audio_query_{case}.json").read_text(encoding="utf-8")) def _meta() -> dict: return json.loads((FIXTURES / "synth_meta.json").read_text(encoding="utf-8")) def _mora(text: str, vowel: str, vowel_length: float, consonant=None, consonant_length=None): return { "text": text, "vowel": vowel, "vowel_length": vowel_length, "consonant": consonant, "consonant_length": consonant_length, "pitch": 5.5, } def test_vowel_mapping(): """AVTR-02: every VOICEVOX vowel symbol, including the devoiced uppercase ones.""" assert VOWEL_TO_VISEME["a"] == "aa" assert VOWEL_TO_VISEME["i"] == "ih" assert VOWEL_TO_VISEME["u"] == "ou" assert VOWEL_TO_VISEME["e"] == "ee" assert VOWEL_TO_VISEME["o"] == "oh" # Japanese devoices /i/ and /u/ between voiceless consonants constantly - です is # "d e s U", した is "sh I t a". VOICEVOX emits those as UPPERCASE. A lowercase-only # table freezes the mouth on every polite form. for v in "AIUEO": assert VOWEL_TO_VISEME[v] == VOWEL_TO_VISEME[v.lower()], v assert frozenset("AIUEO") == DEVOICED assert VOWEL_TO_VISEME["N"] == "closed" # ん assert VOWEL_TO_VISEME["cl"] == "closed" # っ assert VOWEL_TO_VISEME["pau"] == "closed" # silence # Exactly 13: an extra key is an invented symbol, a missing key is a frozen mouth. assert len(VOWEL_TO_VISEME) == 13, sorted(VOWEL_TO_VISEME) # Devoiced vowels still open the mouth, just less. assert viseme_for("u") == ("ou", 1.0) assert viseme_for("U") == ("ou", 0.5) assert viseme_for("I") == ("ih", 0.5) assert viseme_for("N") == ("closed", 0.0) assert viseme_for("pau") == ("closed", 0.0) # Same claim end-to-end: a one-mora timeline carries the reduced weight through. voiced = build_timeline(_one_mora_query(_mora("ス", "u", 0.1, "s", 0.05))) devoiced = build_timeline(_one_mora_query(_mora("ス", "U", 0.1, "s", 0.05))) assert [(e.viseme, e.weight) for e in voiced if e.weight] == [("ou", 1.0)] assert [(e.viseme, e.weight) for e in devoiced if e.weight] == [("ou", 0.5)] # A new VOICEVOX symbol must fail loudly rather than animate wrongly. with pytest.raises(KeyError): viseme_for("x") with pytest.raises(KeyError): VOWEL_TO_VISEME["q"] @pytest.mark.parametrize(("frames", "expected", "half_up"), BANKERS_VECTORS) def test_frame_quantisation_bankers(frames, expected, half_up): """AVTR-02: round-half-to-even, matching np.round inside VOICEVOX itself. The VOICEVOX source carries the warning 「NOTE: `round` は偶数丸め。移植時に取扱い注意。」 Python's built-in round() is banker's, so a Python port matches for free; JavaScript's Math.round() is round-half-up and would disagree on every exact-half boundary. That is an independent, concrete reason this arithmetic lives in Python. """ assert FRAMERATE == 93.75 assert FRAMERATE == 24000 / 256 seconds = frames / FRAMERATE # Fail loudly if a future Python stops round-tripping, rather than passing by luck. assert seconds * FRAMERATE == frames, (frames, seconds * FRAMERATE) assert to_frame(seconds) == expected # Provably NOT round-half-up: on the exact-half boundaries the two modes disagree. if expected != half_up: assert to_frame(seconds) != math.floor(seconds * FRAMERATE + 0.5) assert math.floor(seconds * FRAMERATE + 0.5) == half_up def _one_mora_query(mora: dict, speed: float = 1.0, pre: float = 0.0, post: float = 0.0) -> dict: return { "accent_phrases": [{"moras": [mora], "accent": 1, "pause_mora": None}], "speedScale": speed, "prePhonemeLength": pre, "postPhonemeLength": post, } def test_pause_mora_ordering(): """AVTR-02: a phrase's pause_mora follows its moras. Inverting it shifts the whole phrase.""" query = { "accent_phrases": [ { "moras": [_mora("ア", "a", 0.10), _mora("キ", "i", 0.10, "k", 0.05)], "accent": 1, "pause_mora": _mora("、", "pau", 0.30), }, { "moras": [_mora("オ", "o", 0.10)], "accent": 1, "pause_mora": None, }, ], "speedScale": 1.0, "prePhonemeLength": 0.0, "postPhonemeLength": 0.0, } events = build_timeline(query) # pre/post silences are zero-length here but still emitted, so the shape is # [pre] a, k, i, pause, o, [post]. assert [e.viseme for e in events] == [ "closed", # prePhonemeLength (0 frames) "aa", # ア "closed", # k "ih", # キ "closed", # the pause_mora - AFTER its phrase, not before "oh", # オ "closed", # postPhonemeLength (0 frames) ] pause = events[4] assert pause.viseme == "closed" assert pause.weight == 0.0 assert pause.dur == frames_for(0.30) / FRAMERATE # The pause sits between キ and オ in time, which is the property the ordering exists for. # Compared in frames, not seconds: `t` comes from an exact integer frame accumulator while # `t + dur` is a float sum, so the two differ in the last ULP. That gap is precisely why the # builder accumulates integers - asserting on the float sum would be asserting on the bug. frames = [(round(e.t * FRAMERATE), round(e.dur * FRAMERATE)) for e in events] assert frames[3][0] + frames[3][1] == frames[4][0] assert frames[4][0] + frames[4][1] == frames[5][0] assert all(a[0] + a[1] == b[0] for a, b in zip(frames, frames[1:], strict=False)) # A pause_mora has no consonant, so it contributes exactly one event. assert sum(1 for e in events if e.viseme == "closed" and e.dur > 0) == 2 # k and the pause def test_pre_post_silence_scaled(): """AVTR-02: the pre/post silences exist, bracket the utterance, and ARE scaled by speed. VOICEVOX inserts them before the speed step, so they are divided like any other phoneme. Leaving them unscaled is Pitfall 5's symptom - normal speed syncs, slow speed drifts. """ for case, speed in (("long", 1.0), ("slow", 0.75)): query = _query(case) assert query["speedScale"] == speed events = build_timeline(query) first, last = events[0], events[-1] assert first.viseme == "closed" and first.weight == 0.0 assert last.viseme == "closed" and last.weight == 0.0 assert first.t == 0.0 assert first.dur == frames_for(query["prePhonemeLength"], speed) / FRAMERATE, case assert last.dur == frames_for(query["postPhonemeLength"], speed) / FRAMERATE, case # The two fixtures share a prePhonemeLength, so the silences are directly comparable: # 0.1 s is 9 frames at speed 1.0 and 12 frames at 0.75. Unscaled silences would be equal. normal = build_timeline(_query("long")) slow = build_timeline(_query("slow")) assert _query("long")["prePhonemeLength"] == _query("slow")["prePhonemeLength"] assert round(normal[0].dur * FRAMERATE) == 9 assert round(slow[0].dur * FRAMERATE) == 12 assert slow[0].dur > normal[0].dur assert slow[-1].dur > normal[-1].dur # Same claim on a synthetic query, free of any fixture-specific coincidence. mora = _mora("ア", "a", 0.2) one = build_timeline(_one_mora_query(mora, speed=1.0, pre=0.1, post=0.1)) half = build_timeline(_one_mora_query(mora, speed=0.5, pre=0.1, post=0.1)) assert round(one[0].dur * FRAMERATE) == 9 assert round(half[0].dur * FRAMERATE) == 18 def test_no_drift_long_utterance(): """AVTR-02: on a 20+ mora sentence the timeline ends where the audio ends, within a frame.""" query = _query("long") meta = _meta()["cases"]["long"] true_duration = meta["duration_seconds"] events = build_timeline(query) assert len(events) >= 30, f"the long fixture yields only {len(events)} events; too short" total = sum(e.dur for e in events) assert abs(total - true_duration) <= ONE_FRAME, ( f"summed timeline {total} vs true WAV duration {true_duration}: " f"{abs(total - true_duration) * 1000:.3f} ms, over the {ONE_FRAME * 1000:.3f} ms budget" ) end = events[-1].t + events[-1].dur assert abs(end - true_duration) <= ONE_FRAME, ( f"accumulated timeline ends at {end}, audio ends at {true_duration}: " f"{abs(end - true_duration) * 1000:.3f} ms of accumulated drift" ) # Negative control: accumulate the raw floats instead of the quantised frames. If this does # NOT differ by more than half a frame, the fixture is too short to be catching anything. speed = query["speedScale"] lengths = [query["prePhonemeLength"], query["postPhonemeLength"]] for phrase in query["accent_phrases"]: for m in [*phrase["moras"], *([phrase["pause_mora"]] if phrase["pause_mora"] else [])]: lengths += [m["vowel_length"]] + ([m["consonant_length"]] if m["consonant"] else []) drifted = sum(x / speed for x in lengths) assert abs(drifted - total) > ONE_FRAME / 2, ( f"the float-accumulating variant differs by only " f"{abs(drifted - total) * FRAMERATE:.4f} frames on this fixture, so the fixture is not " f"stressing the accumulator. Replace it with a longer sentence." ) def test_speed_scale(): """VOIC-03: a 0.75x re-read matches the real slow audio and keeps the same mouth shapes.""" meta = _meta()["cases"] normal = build_timeline(_query("long")) slow = build_timeline(_query("slow")) total_normal = sum(e.dur for e in normal) total_slow = sum(e.dur for e in slow) # The assertion that actually matters for lip-sync: each timeline matches its own WAV. assert abs(total_normal - meta["long"]["duration_seconds"]) <= ONE_FRAME assert abs(total_slow - meta["slow"]["duration_seconds"]) <= ONE_FRAME # DELIBERATE DEVIATION from 01-06-PLAN.md, which specified |total_slow - total_normal/0.75| # <= 2 frames. That is false and was measured to be false: VOICEVOX re-quantises every # phoneme AFTER dividing the frame count by speedScale, so the realised ratio lands near # 1/0.75 rather than on it. Measured 1.341085 vs 1.333333 - a 4-frame gap on this sentence, # reproduced exactly by both the builder and the engine. Pinning the measured ratio is the # honest test; pinning the arithmetic ideal would fail against real audio. # See docs/VOICEVOX-SETUP.md, "Frame quantisation". ratio = total_slow / total_normal assert ratio == pytest.approx(1.341085, abs=1e-6), ratio assert ratio == pytest.approx(1 / 0.75, rel=0.02), ( f"speedScale=0.75 did not lengthen the utterance by roughly 1/0.75x: {ratio}" ) # A slower re-read must not change WHICH mouth shapes appear, only how long they last. assert [e.viseme for e in slow] == [e.viseme for e in normal] assert [e.weight for e in slow] == [e.weight for e in normal] assert all(s.dur >= n.dur for s, n in zip(slow, normal, strict=True)) def test_golden_timeline(): """VOIC-01: byte-stable timeline for a fixed sentence, so a core bump cannot go unnoticed.""" golden = json.loads((FIXTURES / "golden_timeline.json").read_text(encoding="utf-8")) events = build_timeline(_query("short")) REGENERATE = ( "The timeline for こんにちは changed. This is NOT a test to update casually: it means " "voicevox_core produced different mora timings than the committed fixtures, so every " "lip-sync in the app has shifted. Confirm the engine/model versions in " "tests/fixtures/synth_meta.json, then regenerate and REVIEW the diff:\n" " uv run --extra voice python tests/fixtures/make_synth_fixtures.py\n" " uv run python tests/fixtures/make_golden_timeline.py" ) assert len(events) == len(golden), f"{len(events)} events vs {len(golden)} golden\n{REGENERATE}" for i, (event, want) in enumerate(zip(events, golden, strict=True)): # Exact float equality is correct here: every value is a whole frame count divided by # FRAMERATE, so any difference at all is a real timing change. assert event.viseme == want["viseme"], f"event {i}\n{REGENERATE}" assert event.weight == want["weight"], f"event {i}\n{REGENERATE}" assert event.t == want["t"], f"event {i}: t {event.t} vs {want['t']}\n{REGENERATE}" assert event.dur == want["dur"], ( f"event {i}: dur {event.dur} vs {want['dur']}\n{REGENERATE}" ) # こんにちは = k o N n i ch i w a, bracketed by the pre/post silences. assert [e.viseme for e in events] == [ "closed", # prePhonemeLength "closed", # k "oh", # コ "closed", # ン "closed", # n "ih", # ニ "closed", # ch "ih", # チ "closed", # w "aa", # ワ "closed", # postPhonemeLength ] # The transport form rounds t and dur to 6 dp. Sub-microsecond, so it cannot move a boundary # anywhere near the 10.7 ms frame; this pins that claim rather than assuming it. for event, compact in zip(events, timeline_to_dicts(events), strict=True): assert abs(compact["t"] - event.t) <= 1e-6 assert abs(compact["dur"] - event.dur) <= 1e-6 assert compact["viseme"] == event.viseme assert compact["weight"] == event.weight