Spaces:
Running on Zero
Running on Zero
File size: 5,121 Bytes
b0f00f6 122fc80 b0f00f6 122fc80 b0f00f6 122fc80 b0f00f6 122fc80 b0f00f6 122fc80 b0f00f6 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 | """VOIC-01: synthesize() returns real audio plus a parseable AudioQuery, on zero GPU.
The whole module skips where the ``voicevox_core`` wheel is absent so a contributor without it
still gets a green quick loop. The wheel is not on PyPI; see ``docs/VOICEVOX-SETUP.md``.
"""
from __future__ import annotations
import ast
import json
from pathlib import Path
import pytest
pytest.importorskip("voicevox_core")
from japanese_avatar.voice.tts import SPEAKER_STYLE_ID, synthesize # noqa: E402
VOICE_PKG = Path(__file__).resolve().parents[1] / "src" / "japanese_avatar" / "voice"
SHORT_TEXT = "γγγ«γ‘γ―"
LONG_TEXT = "δ»ζ₯γ―γγ倩ζ°γ§γγγγε
¬εγζ£ζ©γγ¦γγγθ²·γη©γ«θ‘γγΎγγγ"
@pytest.fixture(scope="module")
def synth():
"""Memoised synthesis, keyed by (text, speed).
The engine is deterministic, so synthesising the same pair twice is pure waste - and this
module would otherwise do it six times over, pushing the quick loop past the 15 s feedback
budget in ``01-VALIDATION.md``. Three distinct syntheses cover every test below.
"""
cache: dict[tuple[str, float], object] = {}
def _synth(text: str, speed: float = 1.0):
key = (text, speed)
if key not in cache:
cache[key] = synthesize(text, speed=speed)
return cache[key]
return _synth
@pytest.fixture(scope="module")
def short_result(synth):
return synth(SHORT_TEXT)
def test_synthesize_returns_nonempty_wav(short_result):
assert short_result.wav_bytes[:4] == b"RIFF"
assert len(short_result.wav_bytes) > 1000
assert 0.5 < short_result.duration < 3.0
assert short_result.text == SHORT_TEXT
assert short_result.speed_scale == 1.0
def test_audio_query_is_plain_json_roundtrippable(short_result):
query = short_result.audio_query
assert isinstance(query, dict)
assert json.loads(json.dumps(query, ensure_ascii=False)) == query
for key in (
"accent_phrases",
"speedScale",
"prePhonemeLength",
"postPhonemeLength",
"outputSamplingRate",
):
assert key in query, key
def test_moras_carry_timings(synth):
"""Per-mora timings exist, and consonant/consonant_length are genuinely optional.
The optionality is asserted rather than assumed: plan 01-06's timeline builder must not index
into a key that is absent for every vowel-only mora (γ’) and for every pause.
"""
result = synth(LONG_TEXT)
moras = []
for phrase in result.audio_query["accent_phrases"]:
moras.extend(phrase["moras"])
if phrase["pause_mora"]:
moras.append(phrase["pause_mora"])
assert moras, "no moras produced"
for mora in moras:
assert isinstance(mora["vowel"], str) and mora["vowel"]
assert isinstance(mora["vowel_length"], (int, float))
with_consonant = [m for m in moras if m.get("consonant") is not None]
without_consonant = [m for m in moras if m.get("consonant") is None]
assert with_consonant, "expected at least one mora with a consonant"
assert all(isinstance(m["consonant_length"], (int, float)) for m in with_consonant)
assert without_consonant, "expected at least one mora with no consonant at all"
def test_speed_scale_lengthens_audio(synth):
normal = synth(LONG_TEXT, speed=1.0)
slow = synth(LONG_TEXT, speed=0.75)
assert slow.duration > normal.duration
ratio = slow.duration / normal.duration
assert abs(ratio - 1 / 0.75) < 0.02 * (1 / 0.75), ratio
assert slow.audio_query["speedScale"] == 0.75
def test_output_sampling_rate_is_24000(short_result):
"""93.75 fps in plan 01-06 is exactly 24000 / 256; the rate is load-bearing, not incidental."""
assert short_result.audio_query["outputSamplingRate"] == 24000
assert SPEAKER_STYLE_ID == 3
def test_fixtures_match_current_engine(synth_meta, synth):
"""Regression guard: if a voicevox_core bump changes timings, this fails loudly
rather than test_golden_timeline failing mysteriously in plan 01-06."""
for case, meta in synth_meta["cases"].items():
r = synth(meta["text"], speed=meta["speed_scale"])
assert abs(r.duration - meta["duration_seconds"]) < 0.011, (
f"{case}: engine now produces {r.duration}s vs recorded {meta['duration_seconds']}s. "
"Regenerate with tests/fixtures/make_synth_fixtures.py and re-review the "
"golden timeline."
)
def test_no_gpu_imports_on_synthesis_path():
modules = sorted(VOICE_PKG.glob("*.py"))
assert modules, f"no modules found under {VOICE_PKG}"
banned = {"spaces", "torch"}
for path in modules:
tree = ast.parse(path.read_text(encoding="utf-8"), filename=str(path))
for node in ast.walk(tree):
if isinstance(node, ast.Import):
for alias in node.names:
assert alias.name.split(".")[0] not in banned, f"{path.name}: {alias.name}"
elif isinstance(node, ast.ImportFrom) and node.module:
assert node.module.split(".")[0] not in banned, f"{path.name}: {node.module}"
|