Spaces:
Running on Zero
Running on Zero
Download tests/test_tts_contract.py from WolfDavid/japanese-learning-avatar: direct link, hf CLI and curl.
- Browser
- Download file 5.12 kB
-
https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/f13c9e2db20eb1318242fc3e3dcebe3a465e367d/tests/test_tts_contract.py
- Command line
-
hf download hf://spaces/WolfDavid/japanese-learning-avatar@f13c9e2db20eb1318242fc3e3dcebe3a465e367d/tests/test_tts_contract.py
-
curl -L -o test_tts_contract.py https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/f13c9e2db20eb1318242fc3e3dcebe3a465e367d/tests/test_tts_contract.py
5.12 kB
| """VOIC-01: synthesize() returns real audio plus a parseable AudioQuery, on zero GPU. | |
| The whole module skips where the ``voicevox_core`` wheel is absent so a contributor without it | |
| still gets a green quick loop. The wheel is not on PyPI; see ``docs/VOICEVOX-SETUP.md``. | |
| """ | |
| from __future__ import annotations | |
| import ast | |
| import json | |
| from pathlib import Path | |
| import pytest | |
| pytest.importorskip("voicevox_core") | |
| from japanese_avatar.voice.tts import SPEAKER_STYLE_ID, synthesize # noqa: E402 | |
| VOICE_PKG = Path(__file__).resolve().parents[1] / "src" / "japanese_avatar" / "voice" | |
| SHORT_TEXT = "γγγ«γ‘γ―" | |
| LONG_TEXT = "δ»ζ₯γ―γγ倩ζ°γ§γγγγε ¬εγζ£ζ©γγ¦γγγθ²·γη©γ«θ‘γγΎγγγ" | |
| def synth(): | |
| """Memoised synthesis, keyed by (text, speed). | |
| The engine is deterministic, so synthesising the same pair twice is pure waste - and this | |
| module would otherwise do it six times over, pushing the quick loop past the 15 s feedback | |
| budget in ``01-VALIDATION.md``. Three distinct syntheses cover every test below. | |
| """ | |
| cache: dict[tuple[str, float], object] = {} | |
| def _synth(text: str, speed: float = 1.0): | |
| key = (text, speed) | |
| if key not in cache: | |
| cache[key] = synthesize(text, speed=speed) | |
| return cache[key] | |
| return _synth | |
| def short_result(synth): | |
| return synth(SHORT_TEXT) | |
| def test_synthesize_returns_nonempty_wav(short_result): | |
| assert short_result.wav_bytes[:4] == b"RIFF" | |
| assert len(short_result.wav_bytes) > 1000 | |
| assert 0.5 < short_result.duration < 3.0 | |
| assert short_result.text == SHORT_TEXT | |
| assert short_result.speed_scale == 1.0 | |
| def test_audio_query_is_plain_json_roundtrippable(short_result): | |
| query = short_result.audio_query | |
| assert isinstance(query, dict) | |
| assert json.loads(json.dumps(query, ensure_ascii=False)) == query | |
| for key in ( | |
| "accent_phrases", | |
| "speedScale", | |
| "prePhonemeLength", | |
| "postPhonemeLength", | |
| "outputSamplingRate", | |
| ): | |
| assert key in query, key | |
| def test_moras_carry_timings(synth): | |
| """Per-mora timings exist, and consonant/consonant_length are genuinely optional. | |
| The optionality is asserted rather than assumed: plan 01-06's timeline builder must not index | |
| into a key that is absent for every vowel-only mora (γ’) and for every pause. | |
| """ | |
| result = synth(LONG_TEXT) | |
| moras = [] | |
| for phrase in result.audio_query["accent_phrases"]: | |
| moras.extend(phrase["moras"]) | |
| if phrase["pause_mora"]: | |
| moras.append(phrase["pause_mora"]) | |
| assert moras, "no moras produced" | |
| for mora in moras: | |
| assert isinstance(mora["vowel"], str) and mora["vowel"] | |
| assert isinstance(mora["vowel_length"], (int, float)) | |
| with_consonant = [m for m in moras if m.get("consonant") is not None] | |
| without_consonant = [m for m in moras if m.get("consonant") is None] | |
| assert with_consonant, "expected at least one mora with a consonant" | |
| assert all(isinstance(m["consonant_length"], (int, float)) for m in with_consonant) | |
| assert without_consonant, "expected at least one mora with no consonant at all" | |
| def test_speed_scale_lengthens_audio(synth): | |
| normal = synth(LONG_TEXT, speed=1.0) | |
| slow = synth(LONG_TEXT, speed=0.75) | |
| assert slow.duration > normal.duration | |
| ratio = slow.duration / normal.duration | |
| assert abs(ratio - 1 / 0.75) < 0.02 * (1 / 0.75), ratio | |
| assert slow.audio_query["speedScale"] == 0.75 | |
| def test_output_sampling_rate_is_24000(short_result): | |
| """93.75 fps in plan 01-06 is exactly 24000 / 256; the rate is load-bearing, not incidental.""" | |
| assert short_result.audio_query["outputSamplingRate"] == 24000 | |
| assert SPEAKER_STYLE_ID == 3 | |
| def test_fixtures_match_current_engine(synth_meta, synth): | |
| """Regression guard: if a voicevox_core bump changes timings, this fails loudly | |
| rather than test_golden_timeline failing mysteriously in plan 01-06.""" | |
| for case, meta in synth_meta["cases"].items(): | |
| r = synth(meta["text"], speed=meta["speed_scale"]) | |
| assert abs(r.duration - meta["duration_seconds"]) < 0.011, ( | |
| f"{case}: engine now produces {r.duration}s vs recorded {meta['duration_seconds']}s. " | |
| "Regenerate with tests/fixtures/make_synth_fixtures.py and re-review the " | |
| "golden timeline." | |
| ) | |
| def test_no_gpu_imports_on_synthesis_path(): | |
| modules = sorted(VOICE_PKG.glob("*.py")) | |
| assert modules, f"no modules found under {VOICE_PKG}" | |
| banned = {"spaces", "torch"} | |
| for path in modules: | |
| tree = ast.parse(path.read_text(encoding="utf-8"), filename=str(path)) | |
| for node in ast.walk(tree): | |
| if isinstance(node, ast.Import): | |
| for alias in node.names: | |
| assert alias.name.split(".")[0] not in banned, f"{path.name}: {alias.name}" | |
| elif isinstance(node, ast.ImportFrom) and node.module: | |
| assert node.module.split(".")[0] not in banned, f"{path.name}: {node.module}" | |