"""VOIC-01: synthesize() returns real audio plus a parseable AudioQuery, on zero GPU. The whole module skips where the ``voicevox_core`` wheel is absent so a contributor without it still gets a green quick loop. The wheel is not on PyPI; see ``docs/VOICEVOX-SETUP.md``. """ from __future__ import annotations import ast import json from pathlib import Path import pytest pytest.importorskip("voicevox_core") from japanese_avatar.voice.tts import SPEAKER_STYLE_ID, synthesize # noqa: E402 VOICE_PKG = Path(__file__).resolve().parents[1] / "src" / "japanese_avatar" / "voice" SHORT_TEXT = "こんにちは" LONG_TEXT = "今日はいい天気ですから、公園を散歩してから、買い物に行きました。" @pytest.fixture(scope="module") def synth(): """Memoised synthesis, keyed by (text, speed). The engine is deterministic, so synthesising the same pair twice is pure waste - and this module would otherwise do it six times over, pushing the quick loop past the 15 s feedback budget in ``01-VALIDATION.md``. Three distinct syntheses cover every test below. """ cache: dict[tuple[str, float], object] = {} def _synth(text: str, speed: float = 1.0): key = (text, speed) if key not in cache: cache[key] = synthesize(text, speed=speed) return cache[key] return _synth @pytest.fixture(scope="module") def short_result(synth): return synth(SHORT_TEXT) def test_synthesize_returns_nonempty_wav(short_result): assert short_result.wav_bytes[:4] == b"RIFF" assert len(short_result.wav_bytes) > 1000 assert 0.5 < short_result.duration < 3.0 assert short_result.text == SHORT_TEXT assert short_result.speed_scale == 1.0 def test_audio_query_is_plain_json_roundtrippable(short_result): query = short_result.audio_query assert isinstance(query, dict) assert json.loads(json.dumps(query, ensure_ascii=False)) == query for key in ( "accent_phrases", "speedScale", "prePhonemeLength", "postPhonemeLength", "outputSamplingRate", ): assert key in query, key def test_moras_carry_timings(synth): """Per-mora timings exist, and consonant/consonant_length are genuinely optional. The optionality is asserted rather than assumed: plan 01-06's timeline builder must not index into a key that is absent for every vowel-only mora (ア) and for every pause. """ result = synth(LONG_TEXT) moras = [] for phrase in result.audio_query["accent_phrases"]: moras.extend(phrase["moras"]) if phrase["pause_mora"]: moras.append(phrase["pause_mora"]) assert moras, "no moras produced" for mora in moras: assert isinstance(mora["vowel"], str) and mora["vowel"] assert isinstance(mora["vowel_length"], (int, float)) with_consonant = [m for m in moras if m.get("consonant") is not None] without_consonant = [m for m in moras if m.get("consonant") is None] assert with_consonant, "expected at least one mora with a consonant" assert all(isinstance(m["consonant_length"], (int, float)) for m in with_consonant) assert without_consonant, "expected at least one mora with no consonant at all" def test_speed_scale_lengthens_audio(synth): normal = synth(LONG_TEXT, speed=1.0) slow = synth(LONG_TEXT, speed=0.75) assert slow.duration > normal.duration ratio = slow.duration / normal.duration assert abs(ratio - 1 / 0.75) < 0.02 * (1 / 0.75), ratio assert slow.audio_query["speedScale"] == 0.75 def test_output_sampling_rate_is_24000(short_result): """93.75 fps in plan 01-06 is exactly 24000 / 256; the rate is load-bearing, not incidental.""" assert short_result.audio_query["outputSamplingRate"] == 24000 assert SPEAKER_STYLE_ID == 3 def test_fixtures_match_current_engine(synth_meta, synth): """Regression guard: if a voicevox_core bump changes timings, this fails loudly rather than test_golden_timeline failing mysteriously in plan 01-06.""" for case, meta in synth_meta["cases"].items(): r = synth(meta["text"], speed=meta["speed_scale"]) assert abs(r.duration - meta["duration_seconds"]) < 0.011, ( f"{case}: engine now produces {r.duration}s vs recorded {meta['duration_seconds']}s. " "Regenerate with tests/fixtures/make_synth_fixtures.py and re-review the " "golden timeline." ) def test_no_gpu_imports_on_synthesis_path(): modules = sorted(VOICE_PKG.glob("*.py")) assert modules, f"no modules found under {VOICE_PKG}" banned = {"spaces", "torch"} for path in modules: tree = ast.parse(path.read_text(encoding="utf-8"), filename=str(path)) for node in ast.walk(tree): if isinstance(node, ast.Import): for alias in node.names: assert alias.name.split(".")[0] not in banned, f"{path.name}: {alias.name}" elif isinstance(node, ast.ImportFrom) and node.module: assert node.module.split(".")[0] not in banned, f"{path.name}: {node.module}"