japanese-learning-avatar / tests /fixtures /make_synth_fixtures.py
WolfDavid's picture
test(01-04): capture ground-truth synthesis fixtures for AVTR-02
122fc80
Raw History Blame
9.12 kB
"""Regenerate the committed VOICEVOX synthesis fixtures.
uv run --extra voice python tests/fixtures/make_synth_fixtures.py
This is a **maintainer action**, not part of CI. The outputs are the ground truth for every
AVTR-02 assertion in plan 01-06, so every duration here is read from the WAV header and never
summed from the query - a summed query is the very thing the no-drift test exists to check.
Three cases, chosen deliberately:
===== ======================= ===== ==============================================================
Case Text Speed Why this text
===== ======================= ===== ==============================================================
short こんにちは 1.0 Contains ``N`` (ん) between vowels - exercises the closed-mouth
path. Doubles as the push-to-talk fixture for VOIC-02.
long (see LONG_TEXT) 1.0 です devoices to ``d e s U`` and した to ``sh I t a``, so the
fixture exercises the uppercase devoiced vowels that a
lowercase-only lookup silently drops. ~36 moras, long enough
that float-accumulation drift would be plainly visible.
slow same as long 0.75 VOIC-03's mechanism; the timeline must be exactly 1/0.75x
longer, pre/post silence included.
===== ======================= ===== ==============================================================
The generator refuses to write a ``long`` fixture that does not actually contain a devoiced vowel,
a pause mora and 20+ moras. A fixture that does not exercise those branches is the wrong fixture,
and finding that out here is much cheaper than finding it out in plan 01-06.
"""
from __future__ import annotations
import datetime as dt
import json
import sys
import wave
from pathlib import Path
FIXTURES = Path(__file__).resolve().parent
REPO_ROOT = FIXTURES.parents[1]
sys.path.insert(0, str(REPO_ROOT / "src"))
SHORT_TEXT = "こんにちは"
LONG_TEXT = "今日はいい天気ですから、公園を散歩してから、買い物に行きました。"
VOICEVOX_CORE_VERSION = "0.17.0"
VOICEVOX_VVM_VERSION = "0.17.0"
DEVOICED_VOWELS = frozenset("AIUEO")
MIN_LONG_MORAS = 20
#: 24000 Hz / 256 samples. Every phoneme is quantised to a whole number of these.
FRAMERATE = 93.75
CASES = [
("short", SHORT_TEXT, 1.0, "speech_ja.wav"),
("long", LONG_TEXT, 1.0, "speech_ja_long.wav"),
("slow", LONG_TEXT, 0.75, "speech_ja_slow.wav"),
]
def flatten_moras(query: dict) -> list[dict]:
"""Every mora in utterance order, each accent phrase's pause mora following its moras."""
moras: list[dict] = []
for phrase in query["accent_phrases"]:
moras.extend(phrase["moras"])
if phrase["pause_mora"]:
moras.append(phrase["pause_mora"])
return moras
def phoneme_lengths(query: dict) -> list[float]:
"""Unscaled phoneme durations in seconds, wrapped in the pre/post silence, in order."""
silence = {"vowel": "pau", "consonant": None, "consonant_length": None}
moras = [
{**silence, "vowel_length": query["prePhonemeLength"]},
*flatten_moras(query),
{**silence, "vowel_length": query["postPhonemeLength"]},
]
lengths: list[float] = []
for mora in moras:
if mora["consonant"] is not None:
lengths.append(mora["consonant_length"])
lengths.append(mora["vowel_length"])
return lengths
def write_wav(path: Path, wav_bytes: bytes) -> tuple[float, int]:
"""Write the WAV; return its true duration and frame count from the header just written.
``frame_count`` is in VOICEVOX frames (256 samples at 24000 Hz = 93.75 fps), not PCM samples,
because that is the unit plan 01-06's +/-1 frame no-drift tolerance is expressed in.
"""
path.write_bytes(wav_bytes)
with wave.open(str(path), "rb") as handle:
assert handle.getframerate() == 24000, handle.getframerate()
assert handle.getnchannels() == 1, handle.getnchannels()
assert handle.getsampwidth() == 2, handle.getsampwidth()
samples = handle.getnframes()
assert samples % 256 == 0, samples
return samples / float(handle.getframerate()), samples // 256
def main() -> int:
try:
import voicevox_core # noqa: F401
except ImportError:
print(
"voicevox_core is not installed, so the fixtures cannot be regenerated.\n"
"Install it with `uv sync --extra dev --extra voice`; see docs/VOICEVOX-SETUP.md.\n"
"Regeneration is a maintainer action - CI does not need it.",
file=sys.stderr,
)
return 1
from japanese_avatar.voice.tts import SPEAKER_STYLE_ID, synthesize
meta: dict = {
"generated": dt.datetime.now(dt.UTC).date().isoformat(),
"voicevox_core": VOICEVOX_CORE_VERSION,
"voicevox_vvm": VOICEVOX_VVM_VERSION,
"style_id": SPEAKER_STYLE_ID,
"cases": {},
}
for case, text, speed, wav_name in CASES:
result = synthesize(text, speed=speed)
query = result.audio_query
query_name = f"audio_query_{case}.json"
(FIXTURES / query_name).write_text(
json.dumps(query, indent=2, ensure_ascii=False, sort_keys=True) + "\n",
encoding="utf-8",
newline="\n",
)
duration, frame_count = write_wav(FIXTURES / wav_name, result.wav_bytes)
# The WAV we just wrote must agree with what synthesize() reported, bit for bit.
assert duration == result.duration, (case, duration, result.duration)
# Ground truth for plan 01-06. See the "Frame quantisation" section of
# docs/VOICEVOX-SETUP.md: quantise at speed 1.0 FIRST, then divide the frame count by
# speedScale and round again. Dividing the length by speedScale before quantising is a
# different, wrong answer that only coincides at speed 1.0.
predicted = sum(
round(round(length * FRAMERATE) / speed) for length in phoneme_lengths(query)
)
assert predicted == frame_count, (case, predicted, frame_count)
moras = flatten_moras(query)
vowels = [m["vowel"] for m in moras]
pause_moras = sum(1 for p in query["accent_phrases"] if p["pause_mora"])
print(
f"{case:>5}: moras={len(moras):<3} pause_moras={pause_moras} frames={frame_count:<5} "
f"duration={duration!r} vowels={sorted(set(vowels))}"
)
if case == "long":
if len(moras) < MIN_LONG_MORAS:
print(
f"FAIL: the long fixture has {len(moras)} moras, fewer than "
f"{MIN_LONG_MORAS}. Pick a longer sentence and record which one.",
file=sys.stderr,
)
return 1
devoiced = sorted(set(vowels) & DEVOICED_VOWELS)
if not devoiced:
print(
"FAIL: the long fixture contains no devoiced vowel (A/I/U/E/O). Its entire "
"job is to exercise that branch - pick a different sentence.",
file=sys.stderr,
)
return 1
if not pause_moras:
print(
"FAIL: the long fixture produced no pause_mora. Pick a sentence with a "
"comma so the pause-ordering branch is covered.",
file=sys.stderr,
)
return 1
print(f" devoiced vowels present: {devoiced}")
meta["cases"][case] = {
"text": text,
"speed_scale": speed,
"query": query_name,
"wav": wav_name,
"duration_seconds": duration,
"frame_count": frame_count,
"mora_count": len(moras),
"pause_mora_count": pause_moras,
"vowel_symbols": vowels,
}
# Not exactly 1/0.75: VOICEVOX re-quantises every phoneme after scaling, so the realised
# ratio lands within a fraction of a percent of it rather than on it. 2% is the tolerance
# the fixture contract is written against.
ratio = meta["cases"]["slow"]["duration_seconds"] / meta["cases"]["long"]["duration_seconds"]
print(f"slow/long duration ratio = {ratio:.6f} (expected ~{1 / 0.75:.6f})")
if abs(ratio - 1 / 0.75) >= 0.02:
print(
f"FAIL: speedScale=0.75 did not lengthen the audio by ~1/0.75x: {ratio}",
file=sys.stderr,
)
return 1
(FIXTURES / "synth_meta.json").write_text(
json.dumps(meta, indent=2, ensure_ascii=False, sort_keys=True) + "\n",
encoding="utf-8",
# Explicit LF: the default translates to CRLF on Windows, which would make the committed
# fixtures differ byte-for-byte depending on which OS last regenerated them.
newline="\n",
)
print(f"wrote {FIXTURES / 'synth_meta.json'}")
return 0
if __name__ == "__main__":
raise SystemExit(main())