"""Regenerate the committed VOICEVOX synthesis fixtures. uv run --extra voice python tests/fixtures/make_synth_fixtures.py This is a **maintainer action**, not part of CI. The outputs are the ground truth for every AVTR-02 assertion in plan 01-06, so every duration here is read from the WAV header and never summed from the query - a summed query is the very thing the no-drift test exists to check. Three cases, chosen deliberately: ===== ======================= ===== ============================================================== Case Text Speed Why this text ===== ======================= ===== ============================================================== short こんにちは 1.0 Contains ``N`` (ん) between vowels - exercises the closed-mouth path. Doubles as the push-to-talk fixture for VOIC-02. long (see LONG_TEXT) 1.0 です devoices to ``d e s U`` and した to ``sh I t a``, so the fixture exercises the uppercase devoiced vowels that a lowercase-only lookup silently drops. ~36 moras, long enough that float-accumulation drift would be plainly visible. slow same as long 0.75 VOIC-03's mechanism; the timeline must be exactly 1/0.75x longer, pre/post silence included. ===== ======================= ===== ============================================================== The generator refuses to write a ``long`` fixture that does not actually contain a devoiced vowel, a pause mora and 20+ moras. A fixture that does not exercise those branches is the wrong fixture, and finding that out here is much cheaper than finding it out in plan 01-06. """ from __future__ import annotations import datetime as dt import json import sys import wave from pathlib import Path FIXTURES = Path(__file__).resolve().parent REPO_ROOT = FIXTURES.parents[1] sys.path.insert(0, str(REPO_ROOT / "src")) SHORT_TEXT = "こんにちは" LONG_TEXT = "今日はいい天気ですから、公園を散歩してから、買い物に行きました。" VOICEVOX_CORE_VERSION = "0.17.0" VOICEVOX_VVM_VERSION = "0.17.0" DEVOICED_VOWELS = frozenset("AIUEO") MIN_LONG_MORAS = 20 #: 24000 Hz / 256 samples. Every phoneme is quantised to a whole number of these. FRAMERATE = 93.75 CASES = [ ("short", SHORT_TEXT, 1.0, "speech_ja.wav"), ("long", LONG_TEXT, 1.0, "speech_ja_long.wav"), ("slow", LONG_TEXT, 0.75, "speech_ja_slow.wav"), ] def flatten_moras(query: dict) -> list[dict]: """Every mora in utterance order, each accent phrase's pause mora following its moras.""" moras: list[dict] = [] for phrase in query["accent_phrases"]: moras.extend(phrase["moras"]) if phrase["pause_mora"]: moras.append(phrase["pause_mora"]) return moras def phoneme_lengths(query: dict) -> list[float]: """Unscaled phoneme durations in seconds, wrapped in the pre/post silence, in order.""" silence = {"vowel": "pau", "consonant": None, "consonant_length": None} moras = [ {**silence, "vowel_length": query["prePhonemeLength"]}, *flatten_moras(query), {**silence, "vowel_length": query["postPhonemeLength"]}, ] lengths: list[float] = [] for mora in moras: if mora["consonant"] is not None: lengths.append(mora["consonant_length"]) lengths.append(mora["vowel_length"]) return lengths def write_wav(path: Path, wav_bytes: bytes) -> tuple[float, int]: """Write the WAV; return its true duration and frame count from the header just written. ``frame_count`` is in VOICEVOX frames (256 samples at 24000 Hz = 93.75 fps), not PCM samples, because that is the unit plan 01-06's +/-1 frame no-drift tolerance is expressed in. """ path.write_bytes(wav_bytes) with wave.open(str(path), "rb") as handle: assert handle.getframerate() == 24000, handle.getframerate() assert handle.getnchannels() == 1, handle.getnchannels() assert handle.getsampwidth() == 2, handle.getsampwidth() samples = handle.getnframes() assert samples % 256 == 0, samples return samples / float(handle.getframerate()), samples // 256 def main() -> int: try: import voicevox_core # noqa: F401 except ImportError: print( "voicevox_core is not installed, so the fixtures cannot be regenerated.\n" "Install it with `uv sync --extra dev --extra voice`; see docs/VOICEVOX-SETUP.md.\n" "Regeneration is a maintainer action - CI does not need it.", file=sys.stderr, ) return 1 from japanese_avatar.voice.tts import SPEAKER_STYLE_ID, synthesize meta: dict = { "generated": dt.datetime.now(dt.UTC).date().isoformat(), "voicevox_core": VOICEVOX_CORE_VERSION, "voicevox_vvm": VOICEVOX_VVM_VERSION, "style_id": SPEAKER_STYLE_ID, "cases": {}, } for case, text, speed, wav_name in CASES: result = synthesize(text, speed=speed) query = result.audio_query query_name = f"audio_query_{case}.json" (FIXTURES / query_name).write_text( json.dumps(query, indent=2, ensure_ascii=False, sort_keys=True) + "\n", encoding="utf-8", newline="\n", ) duration, frame_count = write_wav(FIXTURES / wav_name, result.wav_bytes) # The WAV we just wrote must agree with what synthesize() reported, bit for bit. assert duration == result.duration, (case, duration, result.duration) # Ground truth for plan 01-06. See the "Frame quantisation" section of # docs/VOICEVOX-SETUP.md: quantise at speed 1.0 FIRST, then divide the frame count by # speedScale and round again. Dividing the length by speedScale before quantising is a # different, wrong answer that only coincides at speed 1.0. predicted = sum( round(round(length * FRAMERATE) / speed) for length in phoneme_lengths(query) ) assert predicted == frame_count, (case, predicted, frame_count) moras = flatten_moras(query) vowels = [m["vowel"] for m in moras] pause_moras = sum(1 for p in query["accent_phrases"] if p["pause_mora"]) print( f"{case:>5}: moras={len(moras):<3} pause_moras={pause_moras} frames={frame_count:<5} " f"duration={duration!r} vowels={sorted(set(vowels))}" ) if case == "long": if len(moras) < MIN_LONG_MORAS: print( f"FAIL: the long fixture has {len(moras)} moras, fewer than " f"{MIN_LONG_MORAS}. Pick a longer sentence and record which one.", file=sys.stderr, ) return 1 devoiced = sorted(set(vowels) & DEVOICED_VOWELS) if not devoiced: print( "FAIL: the long fixture contains no devoiced vowel (A/I/U/E/O). Its entire " "job is to exercise that branch - pick a different sentence.", file=sys.stderr, ) return 1 if not pause_moras: print( "FAIL: the long fixture produced no pause_mora. Pick a sentence with a " "comma so the pause-ordering branch is covered.", file=sys.stderr, ) return 1 print(f" devoiced vowels present: {devoiced}") meta["cases"][case] = { "text": text, "speed_scale": speed, "query": query_name, "wav": wav_name, "duration_seconds": duration, "frame_count": frame_count, "mora_count": len(moras), "pause_mora_count": pause_moras, "vowel_symbols": vowels, } # Not exactly 1/0.75: VOICEVOX re-quantises every phoneme after scaling, so the realised # ratio lands within a fraction of a percent of it rather than on it. 2% is the tolerance # the fixture contract is written against. ratio = meta["cases"]["slow"]["duration_seconds"] / meta["cases"]["long"]["duration_seconds"] print(f"slow/long duration ratio = {ratio:.6f} (expected ~{1 / 0.75:.6f})") if abs(ratio - 1 / 0.75) >= 0.02: print( f"FAIL: speedScale=0.75 did not lengthen the audio by ~1/0.75x: {ratio}", file=sys.stderr, ) return 1 (FIXTURES / "synth_meta.json").write_text( json.dumps(meta, indent=2, ensure_ascii=False, sort_keys=True) + "\n", encoding="utf-8", # Explicit LF: the default translates to CRLF on Windows, which would make the committed # fixtures differ byte-for-byte depending on which OS last regenerated them. newline="\n", ) print(f"wrote {FIXTURES / 'synth_meta.json'}") return 0 if __name__ == "__main__": raise SystemExit(main())