Spaces:
Sleeping
Sleeping
File size: 15,385 Bytes
371c5a6 c6dc487 371c5a6 c6dc487 6c967d2 371c5a6 6c967d2 371c5a6 6c967d2 371c5a6 6c967d2 371c5a6 6c967d2 371c5a6 6c967d2 371c5a6 c6dc487 371c5a6 6c967d2 371c5a6 6c967d2 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 350 351 352 353 354 355 356 357 358 359 360 | """AVTR-02 and VOIC-03: the viseme timeline builder.
This is the densest test file in Phase 1 because ``visemes.py`` is the one function whose
failure modes are all silent. A wrong rounding mode, a lowercase-only vowel table or a
float-accumulating loop each produce a timeline that looks entirely plausible in a debugger and
visibly wrong on the avatar's face.
Every fixture here was captured from a real VOICEVOX CORE 0.17.0 synthesis by
``tests/fixtures/make_synth_fixtures.py``; every duration compared against is read from the WAV
header, never summed from the query.
Nothing in this file imports the VOICEVOX wheel - the builder is a pure function over plain
dicts, so the quick loop runs on a machine that has never installed it.
Regenerating the fixtures (a maintainer action, not CI)::
uv run --extra voice python tests/fixtures/make_synth_fixtures.py
uv run python tests/fixtures/make_golden_timeline.py
Run them in that order: the golden timeline is built from ``audio_query_short.json``, so it is
downstream of the synthesis fixtures.
"""
from __future__ import annotations
import json
import math
from pathlib import Path
import pytest
from japanese_avatar.voice.visemes import (
DEVOICED,
FRAMERATE,
VOWEL_TO_VISEME,
build_timeline,
frames_for,
timeline_to_dicts,
to_frame,
viseme_for,
)
FIXTURES = Path(__file__).parent / "fixtures"
#: One VOICEVOX frame. The tolerance every duration assertion in this file is written against.
ONE_FRAME = 1 / FRAMERATE
# frames -> to_frame(frames / FRAMERATE). Round-half-up would give the third column.
# Verified to round-trip exactly through f / 93.75 * 93.75 on CPython.
BANKERS_VECTORS = [
(0.5, 0, 1),
(1.5, 2, 2),
(2.5, 2, 3),
(4.5, 4, 5),
(5.5, 6, 6),
(10.5, 10, 11),
(20.5, 20, 21),
(100.5, 100, 101),
]
def _query(case: str) -> dict:
return json.loads((FIXTURES / f"audio_query_{case}.json").read_text(encoding="utf-8"))
def _meta() -> dict:
return json.loads((FIXTURES / "synth_meta.json").read_text(encoding="utf-8"))
def _mora(text: str, vowel: str, vowel_length: float, consonant=None, consonant_length=None):
return {
"text": text,
"vowel": vowel,
"vowel_length": vowel_length,
"consonant": consonant,
"consonant_length": consonant_length,
"pitch": 5.5,
}
def test_vowel_mapping():
"""AVTR-02: every VOICEVOX vowel symbol, including the devoiced uppercase ones."""
assert VOWEL_TO_VISEME["a"] == "aa"
assert VOWEL_TO_VISEME["i"] == "ih"
assert VOWEL_TO_VISEME["u"] == "ou"
assert VOWEL_TO_VISEME["e"] == "ee"
assert VOWEL_TO_VISEME["o"] == "oh"
# Japanese devoices /i/ and /u/ between voiceless consonants constantly - です is
# "d e s U", した is "sh I t a". VOICEVOX emits those as UPPERCASE. A lowercase-only
# table freezes the mouth on every polite form.
for v in "AIUEO":
assert VOWEL_TO_VISEME[v] == VOWEL_TO_VISEME[v.lower()], v
assert frozenset("AIUEO") == DEVOICED
assert VOWEL_TO_VISEME["N"] == "closed" # ん
assert VOWEL_TO_VISEME["cl"] == "closed" # っ
assert VOWEL_TO_VISEME["pau"] == "closed" # silence
# Exactly 13: an extra key is an invented symbol, a missing key is a frozen mouth.
assert len(VOWEL_TO_VISEME) == 13, sorted(VOWEL_TO_VISEME)
# Devoiced vowels still open the mouth, just less.
assert viseme_for("u") == ("ou", 1.0)
assert viseme_for("U") == ("ou", 0.5)
assert viseme_for("I") == ("ih", 0.5)
assert viseme_for("N") == ("closed", 0.0)
assert viseme_for("pau") == ("closed", 0.0)
# Same claim end-to-end: a one-mora timeline carries the reduced weight through.
voiced = build_timeline(_one_mora_query(_mora("ス", "u", 0.1, "s", 0.05)))
devoiced = build_timeline(_one_mora_query(_mora("ス", "U", 0.1, "s", 0.05)))
assert [(e.viseme, e.weight) for e in voiced if e.weight] == [("ou", 1.0)]
assert [(e.viseme, e.weight) for e in devoiced if e.weight] == [("ou", 0.5)]
# A new VOICEVOX symbol must fail loudly rather than animate wrongly.
with pytest.raises(KeyError):
viseme_for("x")
with pytest.raises(KeyError):
VOWEL_TO_VISEME["q"]
@pytest.mark.parametrize(("frames", "expected", "half_up"), BANKERS_VECTORS)
def test_frame_quantisation_bankers(frames, expected, half_up):
"""AVTR-02: round-half-to-even, matching np.round inside VOICEVOX itself.
The VOICEVOX source carries the warning 「NOTE: `round` は偶数丸め。移植時に取扱い注意。」
Python's built-in round() is banker's, so a Python port matches for free; JavaScript's
Math.round() is round-half-up and would disagree on every exact-half boundary. That is an
independent, concrete reason this arithmetic lives in Python.
"""
assert FRAMERATE == 93.75
assert FRAMERATE == 24000 / 256
seconds = frames / FRAMERATE
# Fail loudly if a future Python stops round-tripping, rather than passing by luck.
assert seconds * FRAMERATE == frames, (frames, seconds * FRAMERATE)
assert to_frame(seconds) == expected
# Provably NOT round-half-up: on the exact-half boundaries the two modes disagree.
if expected != half_up:
assert to_frame(seconds) != math.floor(seconds * FRAMERATE + 0.5)
assert math.floor(seconds * FRAMERATE + 0.5) == half_up
def _one_mora_query(mora: dict, speed: float = 1.0, pre: float = 0.0, post: float = 0.0) -> dict:
return {
"accent_phrases": [{"moras": [mora], "accent": 1, "pause_mora": None}],
"speedScale": speed,
"prePhonemeLength": pre,
"postPhonemeLength": post,
}
def test_pause_mora_ordering():
"""AVTR-02: a phrase's pause_mora follows its moras. Inverting it shifts the whole phrase."""
query = {
"accent_phrases": [
{
"moras": [_mora("ア", "a", 0.10), _mora("キ", "i", 0.10, "k", 0.05)],
"accent": 1,
"pause_mora": _mora("、", "pau", 0.30),
},
{
"moras": [_mora("オ", "o", 0.10)],
"accent": 1,
"pause_mora": None,
},
],
"speedScale": 1.0,
"prePhonemeLength": 0.0,
"postPhonemeLength": 0.0,
}
events = build_timeline(query)
# pre/post silences are zero-length here but still emitted, so the shape is
# [pre] a, k, i, pause, o, [post].
assert [e.viseme for e in events] == [
"closed", # prePhonemeLength (0 frames)
"aa", # ア
"closed", # k
"ih", # キ
"closed", # the pause_mora - AFTER its phrase, not before
"oh", # オ
"closed", # postPhonemeLength (0 frames)
]
pause = events[4]
assert pause.viseme == "closed"
assert pause.weight == 0.0
assert pause.dur == frames_for(0.30) / FRAMERATE
# The pause sits between キ and オ in time, which is the property the ordering exists for.
# Compared in frames, not seconds: `t` comes from an exact integer frame accumulator while
# `t + dur` is a float sum, so the two differ in the last ULP. That gap is precisely why the
# builder accumulates integers - asserting on the float sum would be asserting on the bug.
frames = [(round(e.t * FRAMERATE), round(e.dur * FRAMERATE)) for e in events]
assert frames[3][0] + frames[3][1] == frames[4][0]
assert frames[4][0] + frames[4][1] == frames[5][0]
assert all(a[0] + a[1] == b[0] for a, b in zip(frames, frames[1:], strict=False))
# A pause_mora has no consonant, so it contributes exactly one event.
assert sum(1 for e in events if e.viseme == "closed" and e.dur > 0) == 2 # k and the pause
def test_pre_post_silence_scaled():
"""AVTR-02: the pre/post silences exist, bracket the utterance, and ARE scaled by speed.
VOICEVOX inserts them before the speed step, so they are divided like any other phoneme.
Leaving them unscaled is Pitfall 5's symptom - normal speed syncs, slow speed drifts.
"""
for case, speed in (("long", 1.0), ("slow", 0.75)):
query = _query(case)
assert query["speedScale"] == speed
events = build_timeline(query)
first, last = events[0], events[-1]
assert first.viseme == "closed" and first.weight == 0.0
assert last.viseme == "closed" and last.weight == 0.0
assert first.t == 0.0
assert first.dur == frames_for(query["prePhonemeLength"], speed) / FRAMERATE, case
assert last.dur == frames_for(query["postPhonemeLength"], speed) / FRAMERATE, case
# The two fixtures share a prePhonemeLength, so the silences are directly comparable:
# 0.1 s is 9 frames at speed 1.0 and 12 frames at 0.75. Unscaled silences would be equal.
normal = build_timeline(_query("long"))
slow = build_timeline(_query("slow"))
assert _query("long")["prePhonemeLength"] == _query("slow")["prePhonemeLength"]
assert round(normal[0].dur * FRAMERATE) == 9
assert round(slow[0].dur * FRAMERATE) == 12
assert slow[0].dur > normal[0].dur
assert slow[-1].dur > normal[-1].dur
# Same claim on a synthetic query, free of any fixture-specific coincidence.
mora = _mora("ア", "a", 0.2)
one = build_timeline(_one_mora_query(mora, speed=1.0, pre=0.1, post=0.1))
half = build_timeline(_one_mora_query(mora, speed=0.5, pre=0.1, post=0.1))
assert round(one[0].dur * FRAMERATE) == 9
assert round(half[0].dur * FRAMERATE) == 18
def test_no_drift_long_utterance():
"""AVTR-02: on a 20+ mora sentence the timeline ends where the audio ends, within a frame."""
query = _query("long")
meta = _meta()["cases"]["long"]
true_duration = meta["duration_seconds"]
events = build_timeline(query)
assert len(events) >= 30, f"the long fixture yields only {len(events)} events; too short"
total = sum(e.dur for e in events)
assert abs(total - true_duration) <= ONE_FRAME, (
f"summed timeline {total} vs true WAV duration {true_duration}: "
f"{abs(total - true_duration) * 1000:.3f} ms, over the {ONE_FRAME * 1000:.3f} ms budget"
)
end = events[-1].t + events[-1].dur
assert abs(end - true_duration) <= ONE_FRAME, (
f"accumulated timeline ends at {end}, audio ends at {true_duration}: "
f"{abs(end - true_duration) * 1000:.3f} ms of accumulated drift"
)
# Negative control: accumulate the raw floats instead of the quantised frames. If this does
# NOT differ by more than half a frame, the fixture is too short to be catching anything.
speed = query["speedScale"]
lengths = [query["prePhonemeLength"], query["postPhonemeLength"]]
for phrase in query["accent_phrases"]:
for m in [*phrase["moras"], *([phrase["pause_mora"]] if phrase["pause_mora"] else [])]:
lengths += [m["vowel_length"]] + ([m["consonant_length"]] if m["consonant"] else [])
drifted = sum(x / speed for x in lengths)
assert abs(drifted - total) > ONE_FRAME / 2, (
f"the float-accumulating variant differs by only "
f"{abs(drifted - total) * FRAMERATE:.4f} frames on this fixture, so the fixture is not "
f"stressing the accumulator. Replace it with a longer sentence."
)
def test_speed_scale():
"""VOIC-03: a 0.75x re-read matches the real slow audio and keeps the same mouth shapes."""
meta = _meta()["cases"]
normal = build_timeline(_query("long"))
slow = build_timeline(_query("slow"))
total_normal = sum(e.dur for e in normal)
total_slow = sum(e.dur for e in slow)
# The assertion that actually matters for lip-sync: each timeline matches its own WAV.
assert abs(total_normal - meta["long"]["duration_seconds"]) <= ONE_FRAME
assert abs(total_slow - meta["slow"]["duration_seconds"]) <= ONE_FRAME
# DELIBERATE DEVIATION from 01-06-PLAN.md, which specified |total_slow - total_normal/0.75|
# <= 2 frames. That is false and was measured to be false: VOICEVOX re-quantises every
# phoneme AFTER dividing the frame count by speedScale, so the realised ratio lands near
# 1/0.75 rather than on it. Measured 1.341085 vs 1.333333 - a 4-frame gap on this sentence,
# reproduced exactly by both the builder and the engine. Pinning the measured ratio is the
# honest test; pinning the arithmetic ideal would fail against real audio.
# See docs/VOICEVOX-SETUP.md, "Frame quantisation".
ratio = total_slow / total_normal
assert ratio == pytest.approx(1.341085, abs=1e-6), ratio
assert ratio == pytest.approx(1 / 0.75, rel=0.02), (
f"speedScale=0.75 did not lengthen the utterance by roughly 1/0.75x: {ratio}"
)
# A slower re-read must not change WHICH mouth shapes appear, only how long they last.
assert [e.viseme for e in slow] == [e.viseme for e in normal]
assert [e.weight for e in slow] == [e.weight for e in normal]
assert all(s.dur >= n.dur for s, n in zip(slow, normal, strict=True))
def test_golden_timeline():
"""VOIC-01: byte-stable timeline for a fixed sentence, so a core bump cannot go unnoticed."""
golden = json.loads((FIXTURES / "golden_timeline.json").read_text(encoding="utf-8"))
events = build_timeline(_query("short"))
REGENERATE = (
"The timeline for こんにちは changed. This is NOT a test to update casually: it means "
"voicevox_core produced different mora timings than the committed fixtures, so every "
"lip-sync in the app has shifted. Confirm the engine/model versions in "
"tests/fixtures/synth_meta.json, then regenerate and REVIEW the diff:\n"
" uv run --extra voice python tests/fixtures/make_synth_fixtures.py\n"
" uv run python tests/fixtures/make_golden_timeline.py"
)
assert len(events) == len(golden), f"{len(events)} events vs {len(golden)} golden\n{REGENERATE}"
for i, (event, want) in enumerate(zip(events, golden, strict=True)):
# Exact float equality is correct here: every value is a whole frame count divided by
# FRAMERATE, so any difference at all is a real timing change.
assert event.viseme == want["viseme"], f"event {i}\n{REGENERATE}"
assert event.weight == want["weight"], f"event {i}\n{REGENERATE}"
assert event.t == want["t"], f"event {i}: t {event.t} vs {want['t']}\n{REGENERATE}"
assert event.dur == want["dur"], (
f"event {i}: dur {event.dur} vs {want['dur']}\n{REGENERATE}"
)
# こんにちは = k o N n i ch i w a, bracketed by the pre/post silences.
assert [e.viseme for e in events] == [
"closed", # prePhonemeLength
"closed", # k
"oh", # コ
"closed", # ン
"closed", # n
"ih", # ニ
"closed", # ch
"ih", # チ
"closed", # w
"aa", # ワ
"closed", # postPhonemeLength
]
# The transport form rounds t and dur to 6 dp. Sub-microsecond, so it cannot move a boundary
# anywhere near the 10.7 ms frame; this pins that claim rather than assuming it.
for event, compact in zip(events, timeline_to_dicts(events), strict=True):
assert abs(compact["t"] - event.t) <= 1e-6
assert abs(compact["dur"] - event.dur) <= 1e-6
assert compact["viseme"] == event.viseme
assert compact["weight"] == event.weight
|