WolfDavid's picture
test(01-02): add deterministic silence and cafe-noise audio fixtures
4a095fc
Raw History Blame
1.89 kB
"""Generate the non-speech audio fixtures. Deterministic: fixed seed, fixed params.
Chromium's --use-file-for-fake-audio-capture requires 16-bit PCM WAV.
VOICEVOX outputs 24000 Hz mono, so match it for consistency across fixtures.
"""
import wave
from pathlib import Path
import numpy as np
SR = 24000
SECONDS = 30
HERE = Path(__file__).parent
def _write(path: Path, samples: np.ndarray) -> None:
pcm = np.clip(samples, -1.0, 1.0)
pcm = (pcm * 32767).astype("<i2")
with wave.open(str(path), "wb") as w:
w.setnchannels(1)
w.setsampwidth(2)
w.setframerate(SR)
w.writeframes(pcm.tobytes())
def make_silence() -> Path:
p = HERE / "silence_30s.wav"
_write(p, np.zeros(SR * SECONDS, dtype=np.float64))
return p
def make_cafe_noise() -> Path:
"""Pink-ish noise at conversational level: broadband, speech-shaped, no speech.
Deliberately loud (measured RMS 0.0577, i.e. -24.8 dBFS; peak 0.1357). It must be
well ABOVE any plausible RMS floor, so
that plan 01-07's gate is forced to discriminate speech from noise on envelope
modulation rather than on loudness. A quiet noise fixture would let a naive
RMS-only gate pass the test for the wrong reason.
"""
rng = np.random.default_rng(20260826)
white = rng.standard_normal(SR * SECONDS)
pink = np.cumsum(white)
pink = pink - np.mean(pink)
pink = pink / (np.max(np.abs(pink)) or 1.0)
mixed = 0.7 * pink + 0.3 * (white / (np.max(np.abs(white)) or 1.0))
p = HERE / "cafe_noise_30s.wav"
_write(p, mixed * 0.15)
return p
def wav_duration_seconds(path: Path) -> float:
with wave.open(str(path), "rb") as w:
return w.getnframes() / float(w.getframerate())
if __name__ == "__main__":
for fn in (make_silence, make_cafe_noise):
out = fn()
print(f"{out.name}: {wav_duration_seconds(out):.3f}s")