japanese-learning-avatar / tests /test_tts_contract.py
WolfDavid's picture
test(01-04): capture ground-truth synthesis fixtures for AVTR-02
122fc80
Raw History Blame
5.12 kB
"""VOIC-01: synthesize() returns real audio plus a parseable AudioQuery, on zero GPU.
The whole module skips where the ``voicevox_core`` wheel is absent so a contributor without it
still gets a green quick loop. The wheel is not on PyPI; see ``docs/VOICEVOX-SETUP.md``.
"""
from __future__ import annotations
import ast
import json
from pathlib import Path
import pytest
pytest.importorskip("voicevox_core")
from japanese_avatar.voice.tts import SPEAKER_STYLE_ID, synthesize # noqa: E402
VOICE_PKG = Path(__file__).resolve().parents[1] / "src" / "japanese_avatar" / "voice"
SHORT_TEXT = "こんにけは"
LONG_TEXT = "今ζ—₯γ―γ„γ„ε€©ζ°—γ§γ™γ‹γ‚‰γ€ε…¬εœ’γ‚’ζ•£ζ­©γ—γ¦γ‹γ‚‰γ€θ²·γ„η‰©γ«θ‘ŒγγΎγ—γŸγ€‚"
@pytest.fixture(scope="module")
def synth():
"""Memoised synthesis, keyed by (text, speed).
The engine is deterministic, so synthesising the same pair twice is pure waste - and this
module would otherwise do it six times over, pushing the quick loop past the 15 s feedback
budget in ``01-VALIDATION.md``. Three distinct syntheses cover every test below.
"""
cache: dict[tuple[str, float], object] = {}
def _synth(text: str, speed: float = 1.0):
key = (text, speed)
if key not in cache:
cache[key] = synthesize(text, speed=speed)
return cache[key]
return _synth
@pytest.fixture(scope="module")
def short_result(synth):
return synth(SHORT_TEXT)
def test_synthesize_returns_nonempty_wav(short_result):
assert short_result.wav_bytes[:4] == b"RIFF"
assert len(short_result.wav_bytes) > 1000
assert 0.5 < short_result.duration < 3.0
assert short_result.text == SHORT_TEXT
assert short_result.speed_scale == 1.0
def test_audio_query_is_plain_json_roundtrippable(short_result):
query = short_result.audio_query
assert isinstance(query, dict)
assert json.loads(json.dumps(query, ensure_ascii=False)) == query
for key in (
"accent_phrases",
"speedScale",
"prePhonemeLength",
"postPhonemeLength",
"outputSamplingRate",
):
assert key in query, key
def test_moras_carry_timings(synth):
"""Per-mora timings exist, and consonant/consonant_length are genuinely optional.
The optionality is asserted rather than assumed: plan 01-06's timeline builder must not index
into a key that is absent for every vowel-only mora (γ‚’) and for every pause.
"""
result = synth(LONG_TEXT)
moras = []
for phrase in result.audio_query["accent_phrases"]:
moras.extend(phrase["moras"])
if phrase["pause_mora"]:
moras.append(phrase["pause_mora"])
assert moras, "no moras produced"
for mora in moras:
assert isinstance(mora["vowel"], str) and mora["vowel"]
assert isinstance(mora["vowel_length"], (int, float))
with_consonant = [m for m in moras if m.get("consonant") is not None]
without_consonant = [m for m in moras if m.get("consonant") is None]
assert with_consonant, "expected at least one mora with a consonant"
assert all(isinstance(m["consonant_length"], (int, float)) for m in with_consonant)
assert without_consonant, "expected at least one mora with no consonant at all"
def test_speed_scale_lengthens_audio(synth):
normal = synth(LONG_TEXT, speed=1.0)
slow = synth(LONG_TEXT, speed=0.75)
assert slow.duration > normal.duration
ratio = slow.duration / normal.duration
assert abs(ratio - 1 / 0.75) < 0.02 * (1 / 0.75), ratio
assert slow.audio_query["speedScale"] == 0.75
def test_output_sampling_rate_is_24000(short_result):
"""93.75 fps in plan 01-06 is exactly 24000 / 256; the rate is load-bearing, not incidental."""
assert short_result.audio_query["outputSamplingRate"] == 24000
assert SPEAKER_STYLE_ID == 3
def test_fixtures_match_current_engine(synth_meta, synth):
"""Regression guard: if a voicevox_core bump changes timings, this fails loudly
rather than test_golden_timeline failing mysteriously in plan 01-06."""
for case, meta in synth_meta["cases"].items():
r = synth(meta["text"], speed=meta["speed_scale"])
assert abs(r.duration - meta["duration_seconds"]) < 0.011, (
f"{case}: engine now produces {r.duration}s vs recorded {meta['duration_seconds']}s. "
"Regenerate with tests/fixtures/make_synth_fixtures.py and re-review the "
"golden timeline."
)
def test_no_gpu_imports_on_synthesis_path():
modules = sorted(VOICE_PKG.glob("*.py"))
assert modules, f"no modules found under {VOICE_PKG}"
banned = {"spaces", "torch"}
for path in modules:
tree = ast.parse(path.read_text(encoding="utf-8"), filename=str(path))
for node in ast.walk(tree):
if isinstance(node, ast.Import):
for alias in node.names:
assert alias.name.split(".")[0] not in banned, f"{path.name}: {alias.name}"
elif isinstance(node, ast.ImportFrom) and node.module:
assert node.module.split(".")[0] not in banned, f"{path.name}: {node.module}"