Spaces:
Running on CPU Upgrade
Running on CPU Upgrade
Download audio_tools.py from ai-coustics/VoiceFocus: direct link, hf CLI and curl.
- Browser
- Download file 3.08 kB
-
https://huggingface.co/spaces/ai-coustics/VoiceFocus/resolve/fe2c42f6fe32cfbedee3a78caf5bfba928bec004/audio_tools.py
- Command line
-
hf download hf://spaces/ai-coustics/VoiceFocus@fe2c42f6fe32cfbedee3a78caf5bfba928bec004/audio_tools.py
-
curl -L -o audio_tools.py https://huggingface.co/spaces/ai-coustics/VoiceFocus/resolve/fe2c42f6fe32cfbedee3a78caf5bfba928bec004/audio_tools.py
3.08 kB
| from typing import Optional | |
| import numpy as np | |
| import librosa | |
| from PIL import Image | |
| import io | |
| import matplotlib.pyplot as plt | |
| import soundfile as sf | |
| def spec_image( | |
| audio_wav: str, | |
| n_fft: int = 2048, | |
| hop_length: int = 512, | |
| n_mels: int = 128, | |
| fmax: Optional[float] = None, | |
| ) -> Image.Image: | |
| """ | |
| Generate a mel-spectrogram image from an audio file. | |
| """ | |
| y, sr = librosa.load(audio_wav, mono=True, sr=None) | |
| S = librosa.feature.melspectrogram( | |
| y=y, | |
| sr=sr, | |
| n_fft=n_fft, | |
| hop_length=hop_length, | |
| n_mels=n_mels, | |
| fmax=fmax or sr // 2, | |
| ) | |
| S_db = librosa.power_to_db(S, ref=np.max(S)) | |
| fig, ax = plt.subplots(figsize=(8, 3), dpi=150) | |
| img = librosa.display.specshow( | |
| S_db, sr=sr, hop_length=hop_length, x_axis="time", y_axis="mel", ax=ax | |
| ) | |
| cbar = fig.colorbar(img, ax=ax, format="%+2.0f dB") | |
| cbar.set_label("dB") | |
| ax.set_title("Mel-spectrogram") | |
| ax.set_xlabel("Time in s") | |
| ax.set_ylabel("Frequency in Hz") | |
| fig.tight_layout(pad=0.2) | |
| buf = io.BytesIO() | |
| fig.savefig(buf, format="png", bbox_inches="tight", pad_inches=0) | |
| plt.close(fig) | |
| buf.seek(0) | |
| return Image.open(buf).convert("RGB") | |
| def mix_at_snr( | |
| signal_path: str, | |
| noise_path: str, | |
| output_path: str = "output.wav", | |
| snr_db: float = 10.0, | |
| rng: Optional[np.random.Generator] = None, | |
| ) -> bool: | |
| """ | |
| Mix noise into clean audio at a target SNR (in dB). | |
| Args: | |
| signal_wav: Path to clean/foreground audio (wav/mp3). | |
| noise_wav: Path to noise audio (wav/mp3). | |
| snr_db: Desired SNR in dB (signal/noise). | |
| normalize: If True, peak-normalize the mixture to |x|max=1 after mixing | |
| (note: can slightly alter achieved SNR). | |
| rng: Optional numpy Generator for reproducible random cropping. | |
| Returns: | |
| Bool wether clipping occurred. | |
| """ | |
| clipped = False | |
| rng = rng or np.random.default_rng() | |
| sig, sr_s = librosa.load(signal_path, mono=True, sr=None) | |
| noise, sr_n = librosa.load(noise_path, mono=True, sr=None) | |
| # Resample noise if needed | |
| if sr_s != sr_n: | |
| noise = librosa.resample(noise, orig_sr=sr_n, target_sr=sr_s, res_type="kaiser_best") | |
| # Match lengths: | |
| L = len(sig) | |
| if len(noise) < L: | |
| reps = int(np.ceil(L / len(noise))) | |
| noise = np.tile(noise, reps)[:L] | |
| else: | |
| start = rng.integers(0, len(noise) - L + 1) if len(noise) > L else 0 | |
| noise = noise[start : start + L] | |
| sig_power = float(np.mean(sig**2)) | |
| noise_power = float(np.mean(noise**2)) | |
| if sig_power == 0.0: | |
| out = noise * 0.0 | |
| elif noise_power == 0.0: | |
| out = sig.copy() | |
| else: | |
| target_noise_power = sig_power / (10.0 ** (snr_db / 10.0)) | |
| scale = np.sqrt(target_noise_power / noise_power) | |
| noise_scaled = noise * scale | |
| out = sig + noise_scaled | |
| peak = np.max(np.abs(out)) or 1.0 | |
| if peak > 1.0: | |
| clipped = True | |
| out = out / peak | |
| sf.write(output_path, out, sr_s) | |
| return clipped | |