Spaces:
Running on Zero
Running on Zero
Download tests/fixtures/make_synth_fixtures.py from WolfDavid/japanese-learning-avatar: direct link, hf CLI and curl.
- Browser
- Download file 9.12 kB
-
https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/f108e4c8cd374e7e9f66d39cdea425d0f7e2ff6e/tests/fixtures/make_synth_fixtures.py
- Command line
-
hf download hf://spaces/WolfDavid/japanese-learning-avatar@f108e4c8cd374e7e9f66d39cdea425d0f7e2ff6e/tests/fixtures/make_synth_fixtures.py
-
curl -L -o make_synth_fixtures.py https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/f108e4c8cd374e7e9f66d39cdea425d0f7e2ff6e/tests/fixtures/make_synth_fixtures.py
9.12 kB
| """Regenerate the committed VOICEVOX synthesis fixtures. | |
| uv run --extra voice python tests/fixtures/make_synth_fixtures.py | |
| This is a **maintainer action**, not part of CI. The outputs are the ground truth for every | |
| AVTR-02 assertion in plan 01-06, so every duration here is read from the WAV header and never | |
| summed from the query - a summed query is the very thing the no-drift test exists to check. | |
| Three cases, chosen deliberately: | |
| ===== ======================= ===== ============================================================== | |
| Case Text Speed Why this text | |
| ===== ======================= ===== ============================================================== | |
| short こんにちは 1.0 Contains ``N`` (ん) between vowels - exercises the closed-mouth | |
| path. Doubles as the push-to-talk fixture for VOIC-02. | |
| long (see LONG_TEXT) 1.0 です devoices to ``d e s U`` and した to ``sh I t a``, so the | |
| fixture exercises the uppercase devoiced vowels that a | |
| lowercase-only lookup silently drops. ~36 moras, long enough | |
| that float-accumulation drift would be plainly visible. | |
| slow same as long 0.75 VOIC-03's mechanism; the timeline must be exactly 1/0.75x | |
| longer, pre/post silence included. | |
| ===== ======================= ===== ============================================================== | |
| The generator refuses to write a ``long`` fixture that does not actually contain a devoiced vowel, | |
| a pause mora and 20+ moras. A fixture that does not exercise those branches is the wrong fixture, | |
| and finding that out here is much cheaper than finding it out in plan 01-06. | |
| """ | |
| from __future__ import annotations | |
| import datetime as dt | |
| import json | |
| import sys | |
| import wave | |
| from pathlib import Path | |
| FIXTURES = Path(__file__).resolve().parent | |
| REPO_ROOT = FIXTURES.parents[1] | |
| sys.path.insert(0, str(REPO_ROOT / "src")) | |
| SHORT_TEXT = "こんにちは" | |
| LONG_TEXT = "今日はいい天気ですから、公園を散歩してから、買い物に行きました。" | |
| VOICEVOX_CORE_VERSION = "0.17.0" | |
| VOICEVOX_VVM_VERSION = "0.17.0" | |
| DEVOICED_VOWELS = frozenset("AIUEO") | |
| MIN_LONG_MORAS = 20 | |
| #: 24000 Hz / 256 samples. Every phoneme is quantised to a whole number of these. | |
| FRAMERATE = 93.75 | |
| CASES = [ | |
| ("short", SHORT_TEXT, 1.0, "speech_ja.wav"), | |
| ("long", LONG_TEXT, 1.0, "speech_ja_long.wav"), | |
| ("slow", LONG_TEXT, 0.75, "speech_ja_slow.wav"), | |
| ] | |
| def flatten_moras(query: dict) -> list[dict]: | |
| """Every mora in utterance order, each accent phrase's pause mora following its moras.""" | |
| moras: list[dict] = [] | |
| for phrase in query["accent_phrases"]: | |
| moras.extend(phrase["moras"]) | |
| if phrase["pause_mora"]: | |
| moras.append(phrase["pause_mora"]) | |
| return moras | |
| def phoneme_lengths(query: dict) -> list[float]: | |
| """Unscaled phoneme durations in seconds, wrapped in the pre/post silence, in order.""" | |
| silence = {"vowel": "pau", "consonant": None, "consonant_length": None} | |
| moras = [ | |
| {**silence, "vowel_length": query["prePhonemeLength"]}, | |
| *flatten_moras(query), | |
| {**silence, "vowel_length": query["postPhonemeLength"]}, | |
| ] | |
| lengths: list[float] = [] | |
| for mora in moras: | |
| if mora["consonant"] is not None: | |
| lengths.append(mora["consonant_length"]) | |
| lengths.append(mora["vowel_length"]) | |
| return lengths | |
| def write_wav(path: Path, wav_bytes: bytes) -> tuple[float, int]: | |
| """Write the WAV; return its true duration and frame count from the header just written. | |
| ``frame_count`` is in VOICEVOX frames (256 samples at 24000 Hz = 93.75 fps), not PCM samples, | |
| because that is the unit plan 01-06's +/-1 frame no-drift tolerance is expressed in. | |
| """ | |
| path.write_bytes(wav_bytes) | |
| with wave.open(str(path), "rb") as handle: | |
| assert handle.getframerate() == 24000, handle.getframerate() | |
| assert handle.getnchannels() == 1, handle.getnchannels() | |
| assert handle.getsampwidth() == 2, handle.getsampwidth() | |
| samples = handle.getnframes() | |
| assert samples % 256 == 0, samples | |
| return samples / float(handle.getframerate()), samples // 256 | |
| def main() -> int: | |
| try: | |
| import voicevox_core # noqa: F401 | |
| except ImportError: | |
| print( | |
| "voicevox_core is not installed, so the fixtures cannot be regenerated.\n" | |
| "Install it with `uv sync --extra dev --extra voice`; see docs/VOICEVOX-SETUP.md.\n" | |
| "Regeneration is a maintainer action - CI does not need it.", | |
| file=sys.stderr, | |
| ) | |
| return 1 | |
| from japanese_avatar.voice.tts import SPEAKER_STYLE_ID, synthesize | |
| meta: dict = { | |
| "generated": dt.datetime.now(dt.UTC).date().isoformat(), | |
| "voicevox_core": VOICEVOX_CORE_VERSION, | |
| "voicevox_vvm": VOICEVOX_VVM_VERSION, | |
| "style_id": SPEAKER_STYLE_ID, | |
| "cases": {}, | |
| } | |
| for case, text, speed, wav_name in CASES: | |
| result = synthesize(text, speed=speed) | |
| query = result.audio_query | |
| query_name = f"audio_query_{case}.json" | |
| (FIXTURES / query_name).write_text( | |
| json.dumps(query, indent=2, ensure_ascii=False, sort_keys=True) + "\n", | |
| encoding="utf-8", | |
| newline="\n", | |
| ) | |
| duration, frame_count = write_wav(FIXTURES / wav_name, result.wav_bytes) | |
| # The WAV we just wrote must agree with what synthesize() reported, bit for bit. | |
| assert duration == result.duration, (case, duration, result.duration) | |
| # Ground truth for plan 01-06. See the "Frame quantisation" section of | |
| # docs/VOICEVOX-SETUP.md: quantise at speed 1.0 FIRST, then divide the frame count by | |
| # speedScale and round again. Dividing the length by speedScale before quantising is a | |
| # different, wrong answer that only coincides at speed 1.0. | |
| predicted = sum( | |
| round(round(length * FRAMERATE) / speed) for length in phoneme_lengths(query) | |
| ) | |
| assert predicted == frame_count, (case, predicted, frame_count) | |
| moras = flatten_moras(query) | |
| vowels = [m["vowel"] for m in moras] | |
| pause_moras = sum(1 for p in query["accent_phrases"] if p["pause_mora"]) | |
| print( | |
| f"{case:>5}: moras={len(moras):<3} pause_moras={pause_moras} frames={frame_count:<5} " | |
| f"duration={duration!r} vowels={sorted(set(vowels))}" | |
| ) | |
| if case == "long": | |
| if len(moras) < MIN_LONG_MORAS: | |
| print( | |
| f"FAIL: the long fixture has {len(moras)} moras, fewer than " | |
| f"{MIN_LONG_MORAS}. Pick a longer sentence and record which one.", | |
| file=sys.stderr, | |
| ) | |
| return 1 | |
| devoiced = sorted(set(vowels) & DEVOICED_VOWELS) | |
| if not devoiced: | |
| print( | |
| "FAIL: the long fixture contains no devoiced vowel (A/I/U/E/O). Its entire " | |
| "job is to exercise that branch - pick a different sentence.", | |
| file=sys.stderr, | |
| ) | |
| return 1 | |
| if not pause_moras: | |
| print( | |
| "FAIL: the long fixture produced no pause_mora. Pick a sentence with a " | |
| "comma so the pause-ordering branch is covered.", | |
| file=sys.stderr, | |
| ) | |
| return 1 | |
| print(f" devoiced vowels present: {devoiced}") | |
| meta["cases"][case] = { | |
| "text": text, | |
| "speed_scale": speed, | |
| "query": query_name, | |
| "wav": wav_name, | |
| "duration_seconds": duration, | |
| "frame_count": frame_count, | |
| "mora_count": len(moras), | |
| "pause_mora_count": pause_moras, | |
| "vowel_symbols": vowels, | |
| } | |
| # Not exactly 1/0.75: VOICEVOX re-quantises every phoneme after scaling, so the realised | |
| # ratio lands within a fraction of a percent of it rather than on it. 2% is the tolerance | |
| # the fixture contract is written against. | |
| ratio = meta["cases"]["slow"]["duration_seconds"] / meta["cases"]["long"]["duration_seconds"] | |
| print(f"slow/long duration ratio = {ratio:.6f} (expected ~{1 / 0.75:.6f})") | |
| if abs(ratio - 1 / 0.75) >= 0.02: | |
| print( | |
| f"FAIL: speedScale=0.75 did not lengthen the audio by ~1/0.75x: {ratio}", | |
| file=sys.stderr, | |
| ) | |
| return 1 | |
| (FIXTURES / "synth_meta.json").write_text( | |
| json.dumps(meta, indent=2, ensure_ascii=False, sort_keys=True) + "\n", | |
| encoding="utf-8", | |
| # Explicit LF: the default translates to CRLF on Windows, which would make the committed | |
| # fixtures differ byte-for-byte depending on which OS last regenerated them. | |
| newline="\n", | |
| ) | |
| print(f"wrote {FIXTURES / 'synth_meta.json'}") | |
| return 0 | |
| if __name__ == "__main__": | |
| raise SystemExit(main()) | |