Spaces:
Running on Zero
Running on Zero
Download tests/test_visemes.py from WolfDavid/japanese-learning-avatar: direct link, hf CLI and curl.
- Browser
- Download file 15.4 kB
-
https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/706670a3ece0683b9770f0b7db36e4daa955f91e/tests/test_visemes.py
- Command line
-
hf download hf://spaces/WolfDavid/japanese-learning-avatar@706670a3ece0683b9770f0b7db36e4daa955f91e/tests/test_visemes.py
-
curl -L -o test_visemes.py https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/706670a3ece0683b9770f0b7db36e4daa955f91e/tests/test_visemes.py
15.4 kB
| """AVTR-02 and VOIC-03: the viseme timeline builder. | |
| This is the densest test file in Phase 1 because ``visemes.py`` is the one function whose | |
| failure modes are all silent. A wrong rounding mode, a lowercase-only vowel table or a | |
| float-accumulating loop each produce a timeline that looks entirely plausible in a debugger and | |
| visibly wrong on the avatar's face. | |
| Every fixture here was captured from a real VOICEVOX CORE 0.17.0 synthesis by | |
| ``tests/fixtures/make_synth_fixtures.py``; every duration compared against is read from the WAV | |
| header, never summed from the query. | |
| Nothing in this file imports the VOICEVOX wheel - the builder is a pure function over plain | |
| dicts, so the quick loop runs on a machine that has never installed it. | |
| Regenerating the fixtures (a maintainer action, not CI):: | |
| uv run --extra voice python tests/fixtures/make_synth_fixtures.py | |
| uv run python tests/fixtures/make_golden_timeline.py | |
| Run them in that order: the golden timeline is built from ``audio_query_short.json``, so it is | |
| downstream of the synthesis fixtures. | |
| """ | |
| from __future__ import annotations | |
| import json | |
| import math | |
| from pathlib import Path | |
| import pytest | |
| from japanese_avatar.voice.visemes import ( | |
| DEVOICED, | |
| FRAMERATE, | |
| VOWEL_TO_VISEME, | |
| build_timeline, | |
| frames_for, | |
| timeline_to_dicts, | |
| to_frame, | |
| viseme_for, | |
| ) | |
| FIXTURES = Path(__file__).parent / "fixtures" | |
| #: One VOICEVOX frame. The tolerance every duration assertion in this file is written against. | |
| ONE_FRAME = 1 / FRAMERATE | |
| # frames -> to_frame(frames / FRAMERATE). Round-half-up would give the third column. | |
| # Verified to round-trip exactly through f / 93.75 * 93.75 on CPython. | |
| BANKERS_VECTORS = [ | |
| (0.5, 0, 1), | |
| (1.5, 2, 2), | |
| (2.5, 2, 3), | |
| (4.5, 4, 5), | |
| (5.5, 6, 6), | |
| (10.5, 10, 11), | |
| (20.5, 20, 21), | |
| (100.5, 100, 101), | |
| ] | |
| def _query(case: str) -> dict: | |
| return json.loads((FIXTURES / f"audio_query_{case}.json").read_text(encoding="utf-8")) | |
| def _meta() -> dict: | |
| return json.loads((FIXTURES / "synth_meta.json").read_text(encoding="utf-8")) | |
| def _mora(text: str, vowel: str, vowel_length: float, consonant=None, consonant_length=None): | |
| return { | |
| "text": text, | |
| "vowel": vowel, | |
| "vowel_length": vowel_length, | |
| "consonant": consonant, | |
| "consonant_length": consonant_length, | |
| "pitch": 5.5, | |
| } | |
| def test_vowel_mapping(): | |
| """AVTR-02: every VOICEVOX vowel symbol, including the devoiced uppercase ones.""" | |
| assert VOWEL_TO_VISEME["a"] == "aa" | |
| assert VOWEL_TO_VISEME["i"] == "ih" | |
| assert VOWEL_TO_VISEME["u"] == "ou" | |
| assert VOWEL_TO_VISEME["e"] == "ee" | |
| assert VOWEL_TO_VISEME["o"] == "oh" | |
| # Japanese devoices /i/ and /u/ between voiceless consonants constantly - です is | |
| # "d e s U", した is "sh I t a". VOICEVOX emits those as UPPERCASE. A lowercase-only | |
| # table freezes the mouth on every polite form. | |
| for v in "AIUEO": | |
| assert VOWEL_TO_VISEME[v] == VOWEL_TO_VISEME[v.lower()], v | |
| assert frozenset("AIUEO") == DEVOICED | |
| assert VOWEL_TO_VISEME["N"] == "closed" # ん | |
| assert VOWEL_TO_VISEME["cl"] == "closed" # っ | |
| assert VOWEL_TO_VISEME["pau"] == "closed" # silence | |
| # Exactly 13: an extra key is an invented symbol, a missing key is a frozen mouth. | |
| assert len(VOWEL_TO_VISEME) == 13, sorted(VOWEL_TO_VISEME) | |
| # Devoiced vowels still open the mouth, just less. | |
| assert viseme_for("u") == ("ou", 1.0) | |
| assert viseme_for("U") == ("ou", 0.5) | |
| assert viseme_for("I") == ("ih", 0.5) | |
| assert viseme_for("N") == ("closed", 0.0) | |
| assert viseme_for("pau") == ("closed", 0.0) | |
| # Same claim end-to-end: a one-mora timeline carries the reduced weight through. | |
| voiced = build_timeline(_one_mora_query(_mora("ス", "u", 0.1, "s", 0.05))) | |
| devoiced = build_timeline(_one_mora_query(_mora("ス", "U", 0.1, "s", 0.05))) | |
| assert [(e.viseme, e.weight) for e in voiced if e.weight] == [("ou", 1.0)] | |
| assert [(e.viseme, e.weight) for e in devoiced if e.weight] == [("ou", 0.5)] | |
| # A new VOICEVOX symbol must fail loudly rather than animate wrongly. | |
| with pytest.raises(KeyError): | |
| viseme_for("x") | |
| with pytest.raises(KeyError): | |
| VOWEL_TO_VISEME["q"] | |
| def test_frame_quantisation_bankers(frames, expected, half_up): | |
| """AVTR-02: round-half-to-even, matching np.round inside VOICEVOX itself. | |
| The VOICEVOX source carries the warning 「NOTE: `round` は偶数丸め。移植時に取扱い注意。」 | |
| Python's built-in round() is banker's, so a Python port matches for free; JavaScript's | |
| Math.round() is round-half-up and would disagree on every exact-half boundary. That is an | |
| independent, concrete reason this arithmetic lives in Python. | |
| """ | |
| assert FRAMERATE == 93.75 | |
| assert FRAMERATE == 24000 / 256 | |
| seconds = frames / FRAMERATE | |
| # Fail loudly if a future Python stops round-tripping, rather than passing by luck. | |
| assert seconds * FRAMERATE == frames, (frames, seconds * FRAMERATE) | |
| assert to_frame(seconds) == expected | |
| # Provably NOT round-half-up: on the exact-half boundaries the two modes disagree. | |
| if expected != half_up: | |
| assert to_frame(seconds) != math.floor(seconds * FRAMERATE + 0.5) | |
| assert math.floor(seconds * FRAMERATE + 0.5) == half_up | |
| def _one_mora_query(mora: dict, speed: float = 1.0, pre: float = 0.0, post: float = 0.0) -> dict: | |
| return { | |
| "accent_phrases": [{"moras": [mora], "accent": 1, "pause_mora": None}], | |
| "speedScale": speed, | |
| "prePhonemeLength": pre, | |
| "postPhonemeLength": post, | |
| } | |
| def test_pause_mora_ordering(): | |
| """AVTR-02: a phrase's pause_mora follows its moras. Inverting it shifts the whole phrase.""" | |
| query = { | |
| "accent_phrases": [ | |
| { | |
| "moras": [_mora("ア", "a", 0.10), _mora("キ", "i", 0.10, "k", 0.05)], | |
| "accent": 1, | |
| "pause_mora": _mora("、", "pau", 0.30), | |
| }, | |
| { | |
| "moras": [_mora("オ", "o", 0.10)], | |
| "accent": 1, | |
| "pause_mora": None, | |
| }, | |
| ], | |
| "speedScale": 1.0, | |
| "prePhonemeLength": 0.0, | |
| "postPhonemeLength": 0.0, | |
| } | |
| events = build_timeline(query) | |
| # pre/post silences are zero-length here but still emitted, so the shape is | |
| # [pre] a, k, i, pause, o, [post]. | |
| assert [e.viseme for e in events] == [ | |
| "closed", # prePhonemeLength (0 frames) | |
| "aa", # ア | |
| "closed", # k | |
| "ih", # キ | |
| "closed", # the pause_mora - AFTER its phrase, not before | |
| "oh", # オ | |
| "closed", # postPhonemeLength (0 frames) | |
| ] | |
| pause = events[4] | |
| assert pause.viseme == "closed" | |
| assert pause.weight == 0.0 | |
| assert pause.dur == frames_for(0.30) / FRAMERATE | |
| # The pause sits between キ and オ in time, which is the property the ordering exists for. | |
| # Compared in frames, not seconds: `t` comes from an exact integer frame accumulator while | |
| # `t + dur` is a float sum, so the two differ in the last ULP. That gap is precisely why the | |
| # builder accumulates integers - asserting on the float sum would be asserting on the bug. | |
| frames = [(round(e.t * FRAMERATE), round(e.dur * FRAMERATE)) for e in events] | |
| assert frames[3][0] + frames[3][1] == frames[4][0] | |
| assert frames[4][0] + frames[4][1] == frames[5][0] | |
| assert all(a[0] + a[1] == b[0] for a, b in zip(frames, frames[1:], strict=False)) | |
| # A pause_mora has no consonant, so it contributes exactly one event. | |
| assert sum(1 for e in events if e.viseme == "closed" and e.dur > 0) == 2 # k and the pause | |
| def test_pre_post_silence_scaled(): | |
| """AVTR-02: the pre/post silences exist, bracket the utterance, and ARE scaled by speed. | |
| VOICEVOX inserts them before the speed step, so they are divided like any other phoneme. | |
| Leaving them unscaled is Pitfall 5's symptom - normal speed syncs, slow speed drifts. | |
| """ | |
| for case, speed in (("long", 1.0), ("slow", 0.75)): | |
| query = _query(case) | |
| assert query["speedScale"] == speed | |
| events = build_timeline(query) | |
| first, last = events[0], events[-1] | |
| assert first.viseme == "closed" and first.weight == 0.0 | |
| assert last.viseme == "closed" and last.weight == 0.0 | |
| assert first.t == 0.0 | |
| assert first.dur == frames_for(query["prePhonemeLength"], speed) / FRAMERATE, case | |
| assert last.dur == frames_for(query["postPhonemeLength"], speed) / FRAMERATE, case | |
| # The two fixtures share a prePhonemeLength, so the silences are directly comparable: | |
| # 0.1 s is 9 frames at speed 1.0 and 12 frames at 0.75. Unscaled silences would be equal. | |
| normal = build_timeline(_query("long")) | |
| slow = build_timeline(_query("slow")) | |
| assert _query("long")["prePhonemeLength"] == _query("slow")["prePhonemeLength"] | |
| assert round(normal[0].dur * FRAMERATE) == 9 | |
| assert round(slow[0].dur * FRAMERATE) == 12 | |
| assert slow[0].dur > normal[0].dur | |
| assert slow[-1].dur > normal[-1].dur | |
| # Same claim on a synthetic query, free of any fixture-specific coincidence. | |
| mora = _mora("ア", "a", 0.2) | |
| one = build_timeline(_one_mora_query(mora, speed=1.0, pre=0.1, post=0.1)) | |
| half = build_timeline(_one_mora_query(mora, speed=0.5, pre=0.1, post=0.1)) | |
| assert round(one[0].dur * FRAMERATE) == 9 | |
| assert round(half[0].dur * FRAMERATE) == 18 | |
| def test_no_drift_long_utterance(): | |
| """AVTR-02: on a 20+ mora sentence the timeline ends where the audio ends, within a frame.""" | |
| query = _query("long") | |
| meta = _meta()["cases"]["long"] | |
| true_duration = meta["duration_seconds"] | |
| events = build_timeline(query) | |
| assert len(events) >= 30, f"the long fixture yields only {len(events)} events; too short" | |
| total = sum(e.dur for e in events) | |
| assert abs(total - true_duration) <= ONE_FRAME, ( | |
| f"summed timeline {total} vs true WAV duration {true_duration}: " | |
| f"{abs(total - true_duration) * 1000:.3f} ms, over the {ONE_FRAME * 1000:.3f} ms budget" | |
| ) | |
| end = events[-1].t + events[-1].dur | |
| assert abs(end - true_duration) <= ONE_FRAME, ( | |
| f"accumulated timeline ends at {end}, audio ends at {true_duration}: " | |
| f"{abs(end - true_duration) * 1000:.3f} ms of accumulated drift" | |
| ) | |
| # Negative control: accumulate the raw floats instead of the quantised frames. If this does | |
| # NOT differ by more than half a frame, the fixture is too short to be catching anything. | |
| speed = query["speedScale"] | |
| lengths = [query["prePhonemeLength"], query["postPhonemeLength"]] | |
| for phrase in query["accent_phrases"]: | |
| for m in [*phrase["moras"], *([phrase["pause_mora"]] if phrase["pause_mora"] else [])]: | |
| lengths += [m["vowel_length"]] + ([m["consonant_length"]] if m["consonant"] else []) | |
| drifted = sum(x / speed for x in lengths) | |
| assert abs(drifted - total) > ONE_FRAME / 2, ( | |
| f"the float-accumulating variant differs by only " | |
| f"{abs(drifted - total) * FRAMERATE:.4f} frames on this fixture, so the fixture is not " | |
| f"stressing the accumulator. Replace it with a longer sentence." | |
| ) | |
| def test_speed_scale(): | |
| """VOIC-03: a 0.75x re-read matches the real slow audio and keeps the same mouth shapes.""" | |
| meta = _meta()["cases"] | |
| normal = build_timeline(_query("long")) | |
| slow = build_timeline(_query("slow")) | |
| total_normal = sum(e.dur for e in normal) | |
| total_slow = sum(e.dur for e in slow) | |
| # The assertion that actually matters for lip-sync: each timeline matches its own WAV. | |
| assert abs(total_normal - meta["long"]["duration_seconds"]) <= ONE_FRAME | |
| assert abs(total_slow - meta["slow"]["duration_seconds"]) <= ONE_FRAME | |
| # DELIBERATE DEVIATION from 01-06-PLAN.md, which specified |total_slow - total_normal/0.75| | |
| # <= 2 frames. That is false and was measured to be false: VOICEVOX re-quantises every | |
| # phoneme AFTER dividing the frame count by speedScale, so the realised ratio lands near | |
| # 1/0.75 rather than on it. Measured 1.341085 vs 1.333333 - a 4-frame gap on this sentence, | |
| # reproduced exactly by both the builder and the engine. Pinning the measured ratio is the | |
| # honest test; pinning the arithmetic ideal would fail against real audio. | |
| # See docs/VOICEVOX-SETUP.md, "Frame quantisation". | |
| ratio = total_slow / total_normal | |
| assert ratio == pytest.approx(1.341085, abs=1e-6), ratio | |
| assert ratio == pytest.approx(1 / 0.75, rel=0.02), ( | |
| f"speedScale=0.75 did not lengthen the utterance by roughly 1/0.75x: {ratio}" | |
| ) | |
| # A slower re-read must not change WHICH mouth shapes appear, only how long they last. | |
| assert [e.viseme for e in slow] == [e.viseme for e in normal] | |
| assert [e.weight for e in slow] == [e.weight for e in normal] | |
| assert all(s.dur >= n.dur for s, n in zip(slow, normal, strict=True)) | |
| def test_golden_timeline(): | |
| """VOIC-01: byte-stable timeline for a fixed sentence, so a core bump cannot go unnoticed.""" | |
| golden = json.loads((FIXTURES / "golden_timeline.json").read_text(encoding="utf-8")) | |
| events = build_timeline(_query("short")) | |
| REGENERATE = ( | |
| "The timeline for こんにちは changed. This is NOT a test to update casually: it means " | |
| "voicevox_core produced different mora timings than the committed fixtures, so every " | |
| "lip-sync in the app has shifted. Confirm the engine/model versions in " | |
| "tests/fixtures/synth_meta.json, then regenerate and REVIEW the diff:\n" | |
| " uv run --extra voice python tests/fixtures/make_synth_fixtures.py\n" | |
| " uv run python tests/fixtures/make_golden_timeline.py" | |
| ) | |
| assert len(events) == len(golden), f"{len(events)} events vs {len(golden)} golden\n{REGENERATE}" | |
| for i, (event, want) in enumerate(zip(events, golden, strict=True)): | |
| # Exact float equality is correct here: every value is a whole frame count divided by | |
| # FRAMERATE, so any difference at all is a real timing change. | |
| assert event.viseme == want["viseme"], f"event {i}\n{REGENERATE}" | |
| assert event.weight == want["weight"], f"event {i}\n{REGENERATE}" | |
| assert event.t == want["t"], f"event {i}: t {event.t} vs {want['t']}\n{REGENERATE}" | |
| assert event.dur == want["dur"], ( | |
| f"event {i}: dur {event.dur} vs {want['dur']}\n{REGENERATE}" | |
| ) | |
| # こんにちは = k o N n i ch i w a, bracketed by the pre/post silences. | |
| assert [e.viseme for e in events] == [ | |
| "closed", # prePhonemeLength | |
| "closed", # k | |
| "oh", # コ | |
| "closed", # ン | |
| "closed", # n | |
| "ih", # ニ | |
| "closed", # ch | |
| "ih", # チ | |
| "closed", # w | |
| "aa", # ワ | |
| "closed", # postPhonemeLength | |
| ] | |
| # The transport form rounds t and dur to 6 dp. Sub-microsecond, so it cannot move a boundary | |
| # anywhere near the 10.7 ms frame; this pins that claim rather than assuming it. | |
| for event, compact in zip(events, timeline_to_dicts(events), strict=True): | |
| assert abs(compact["t"] - event.t) <= 1e-6 | |
| assert abs(compact["dur"] - event.dur) <= 1e-6 | |
| assert compact["viseme"] == event.viseme | |
| assert compact["weight"] == event.weight | |