File size: 5,121 Bytes
b0f00f6
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
122fc80
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
b0f00f6
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
122fc80
b0f00f6
 
 
 
 
122fc80
b0f00f6
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
122fc80
 
 
b0f00f6
 
 
 
 
 
 
 
 
 
 
 
122fc80
 
 
 
 
 
 
 
 
 
 
 
b0f00f6
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
"""VOIC-01: synthesize() returns real audio plus a parseable AudioQuery, on zero GPU.

The whole module skips where the ``voicevox_core`` wheel is absent so a contributor without it
still gets a green quick loop. The wheel is not on PyPI; see ``docs/VOICEVOX-SETUP.md``.
"""

from __future__ import annotations

import ast
import json
from pathlib import Path

import pytest

pytest.importorskip("voicevox_core")

from japanese_avatar.voice.tts import SPEAKER_STYLE_ID, synthesize  # noqa: E402

VOICE_PKG = Path(__file__).resolve().parents[1] / "src" / "japanese_avatar" / "voice"

SHORT_TEXT = "こんにけは"
LONG_TEXT = "今ζ—₯γ―γ„γ„ε€©ζ°—γ§γ™γ‹γ‚‰γ€ε…¬εœ’γ‚’ζ•£ζ­©γ—γ¦γ‹γ‚‰γ€θ²·γ„η‰©γ«θ‘ŒγγΎγ—γŸγ€‚"


@pytest.fixture(scope="module")
def synth():
    """Memoised synthesis, keyed by (text, speed).

    The engine is deterministic, so synthesising the same pair twice is pure waste - and this
    module would otherwise do it six times over, pushing the quick loop past the 15 s feedback
    budget in ``01-VALIDATION.md``. Three distinct syntheses cover every test below.
    """
    cache: dict[tuple[str, float], object] = {}

    def _synth(text: str, speed: float = 1.0):
        key = (text, speed)
        if key not in cache:
            cache[key] = synthesize(text, speed=speed)
        return cache[key]

    return _synth


@pytest.fixture(scope="module")
def short_result(synth):
    return synth(SHORT_TEXT)


def test_synthesize_returns_nonempty_wav(short_result):
    assert short_result.wav_bytes[:4] == b"RIFF"
    assert len(short_result.wav_bytes) > 1000
    assert 0.5 < short_result.duration < 3.0
    assert short_result.text == SHORT_TEXT
    assert short_result.speed_scale == 1.0


def test_audio_query_is_plain_json_roundtrippable(short_result):
    query = short_result.audio_query
    assert isinstance(query, dict)
    assert json.loads(json.dumps(query, ensure_ascii=False)) == query
    for key in (
        "accent_phrases",
        "speedScale",
        "prePhonemeLength",
        "postPhonemeLength",
        "outputSamplingRate",
    ):
        assert key in query, key


def test_moras_carry_timings(synth):
    """Per-mora timings exist, and consonant/consonant_length are genuinely optional.

    The optionality is asserted rather than assumed: plan 01-06's timeline builder must not index
    into a key that is absent for every vowel-only mora (γ‚’) and for every pause.
    """
    result = synth(LONG_TEXT)
    moras = []
    for phrase in result.audio_query["accent_phrases"]:
        moras.extend(phrase["moras"])
        if phrase["pause_mora"]:
            moras.append(phrase["pause_mora"])

    assert moras, "no moras produced"
    for mora in moras:
        assert isinstance(mora["vowel"], str) and mora["vowel"]
        assert isinstance(mora["vowel_length"], (int, float))

    with_consonant = [m for m in moras if m.get("consonant") is not None]
    without_consonant = [m for m in moras if m.get("consonant") is None]
    assert with_consonant, "expected at least one mora with a consonant"
    assert all(isinstance(m["consonant_length"], (int, float)) for m in with_consonant)
    assert without_consonant, "expected at least one mora with no consonant at all"


def test_speed_scale_lengthens_audio(synth):
    normal = synth(LONG_TEXT, speed=1.0)
    slow = synth(LONG_TEXT, speed=0.75)
    assert slow.duration > normal.duration
    ratio = slow.duration / normal.duration
    assert abs(ratio - 1 / 0.75) < 0.02 * (1 / 0.75), ratio
    assert slow.audio_query["speedScale"] == 0.75


def test_output_sampling_rate_is_24000(short_result):
    """93.75 fps in plan 01-06 is exactly 24000 / 256; the rate is load-bearing, not incidental."""
    assert short_result.audio_query["outputSamplingRate"] == 24000
    assert SPEAKER_STYLE_ID == 3


def test_fixtures_match_current_engine(synth_meta, synth):
    """Regression guard: if a voicevox_core bump changes timings, this fails loudly
    rather than test_golden_timeline failing mysteriously in plan 01-06."""
    for case, meta in synth_meta["cases"].items():
        r = synth(meta["text"], speed=meta["speed_scale"])
        assert abs(r.duration - meta["duration_seconds"]) < 0.011, (
            f"{case}: engine now produces {r.duration}s vs recorded {meta['duration_seconds']}s. "
            "Regenerate with tests/fixtures/make_synth_fixtures.py and re-review the "
            "golden timeline."
        )


def test_no_gpu_imports_on_synthesis_path():
    modules = sorted(VOICE_PKG.glob("*.py"))
    assert modules, f"no modules found under {VOICE_PKG}"
    banned = {"spaces", "torch"}
    for path in modules:
        tree = ast.parse(path.read_text(encoding="utf-8"), filename=str(path))
        for node in ast.walk(tree):
            if isinstance(node, ast.Import):
                for alias in node.names:
                    assert alias.name.split(".")[0] not in banned, f"{path.name}: {alias.name}"
            elif isinstance(node, ast.ImportFrom) and node.module:
                assert node.module.split(".")[0] not in banned, f"{path.name}: {node.module}"