VoiceFocus / audio_tools.py
mariesig
initial demo
25d15ee
Raw History Blame
3.08 kB
from typing import Optional
import numpy as np
import librosa
from PIL import Image
import io
import matplotlib.pyplot as plt
import soundfile as sf
def spec_image(
audio_wav: str,
n_fft: int = 2048,
hop_length: int = 512,
n_mels: int = 128,
fmax: Optional[float] = None,
) -> Image.Image:
"""
Generate a mel-spectrogram image from an audio file.
"""
y, sr = librosa.load(audio_wav, mono=True, sr=None)
S = librosa.feature.melspectrogram(
y=y,
sr=sr,
n_fft=n_fft,
hop_length=hop_length,
n_mels=n_mels,
fmax=fmax or sr // 2,
)
S_db = librosa.power_to_db(S, ref=np.max(S))
fig, ax = plt.subplots(figsize=(8, 3), dpi=150)
img = librosa.display.specshow(
S_db, sr=sr, hop_length=hop_length, x_axis="time", y_axis="mel", ax=ax
)
cbar = fig.colorbar(img, ax=ax, format="%+2.0f dB")
cbar.set_label("dB")
ax.set_title("Mel-spectrogram")
ax.set_xlabel("Time in s")
ax.set_ylabel("Frequency in Hz")
fig.tight_layout(pad=0.2)
buf = io.BytesIO()
fig.savefig(buf, format="png", bbox_inches="tight", pad_inches=0)
plt.close(fig)
buf.seek(0)
return Image.open(buf).convert("RGB")
def mix_at_snr(
signal_path: str,
noise_path: str,
output_path: str = "output.wav",
snr_db: float = 10.0,
rng: Optional[np.random.Generator] = None,
) -> bool:
"""
Mix noise into clean audio at a target SNR (in dB).
Args:
signal_wav: Path to clean/foreground audio (wav/mp3).
noise_wav: Path to noise audio (wav/mp3).
snr_db: Desired SNR in dB (signal/noise).
normalize: If True, peak-normalize the mixture to |x|max=1 after mixing
(note: can slightly alter achieved SNR).
rng: Optional numpy Generator for reproducible random cropping.
Returns:
Bool wether clipping occurred.
"""
clipped = False
rng = rng or np.random.default_rng()
sig, sr_s = librosa.load(signal_path, mono=True, sr=None)
noise, sr_n = librosa.load(noise_path, mono=True, sr=None)
# Resample noise if needed
if sr_s != sr_n:
noise = librosa.resample(noise, orig_sr=sr_n, target_sr=sr_s, res_type="kaiser_best")
# Match lengths:
L = len(sig)
if len(noise) < L:
reps = int(np.ceil(L / len(noise)))
noise = np.tile(noise, reps)[:L]
else:
start = rng.integers(0, len(noise) - L + 1) if len(noise) > L else 0
noise = noise[start : start + L]
sig_power = float(np.mean(sig**2))
noise_power = float(np.mean(noise**2))
if sig_power == 0.0:
out = noise * 0.0
elif noise_power == 0.0:
out = sig.copy()
else:
target_noise_power = sig_power / (10.0 ** (snr_db / 10.0))
scale = np.sqrt(target_noise_power / noise_power)
noise_scaled = noise * scale
out = sig + noise_scaled
peak = np.max(np.abs(out)) or 1.0
if peak > 1.0:
clipped = True
out = out / peak
sf.write(output_path, out, sr_s)
return clipped