Spaces:
Running on CPU Upgrade
Running on CPU Upgrade
Download utils.py from ai-coustics/VoiceFocus: direct link, hf CLI and curl.
- Browser
- Download file 5.11 kB
-
https://huggingface.co/spaces/ai-coustics/VoiceFocus/resolve/ab0481a43d19026f3fd291f05b2b146c857ad746/utils.py
- Command line
-
hf download hf://spaces/ai-coustics/VoiceFocus@ab0481a43d19026f3fd291f05b2b146c857ad746/utils.py
-
curl -L -o utils.py https://huggingface.co/spaces/ai-coustics/VoiceFocus/resolve/ab0481a43d19026f3fd291f05b2b146c857ad746/utils.py
5.11 kB
| from typing import Optional | |
| import numpy as np | |
| import librosa | |
| from PIL import Image | |
| import io | |
| import matplotlib.pyplot as plt | |
| from constants import TARGET_LOUDNESS, TARGET_TP, VAD_OFF, VAD_ON | |
| import pyloudnorm as pyln | |
| import warnings | |
| def get_vad_labels(vad_timestamps: list[list[float]], length: float) -> list[dict]: | |
| subtitles = [] | |
| cur = 0.0 | |
| for start, end in vad_timestamps: | |
| if start > cur: | |
| subtitles.append({ | |
| "text": f"Voice Detection: {VAD_OFF}", | |
| "timestamp": [cur, start] | |
| }) | |
| subtitles.append({ | |
| "text": f"Voice Detection: {VAD_ON}", | |
| "timestamp": [start, end] | |
| }) | |
| cur = end | |
| if cur < length: | |
| subtitles.append({ | |
| "text": f"Voice Detection: {VAD_OFF}", | |
| "timestamp": [cur, length] | |
| }) | |
| return subtitles | |
| def to_gradio_audio(x: np.ndarray, sr: int) -> tuple[int, np.ndarray]: | |
| """Return (sample_rate, int16 mono array) for Gradio Audio. Gradio expects int16; | |
| passing float32 triggers an internal conversion and a warning.""" | |
| x = np.asarray(x) | |
| # Remove extra dims like (1, n, 1) etc. | |
| x = np.squeeze(x) | |
| # If it's (channels, samples), transpose to (samples, channels) | |
| if x.ndim == 2 and x.shape[0] in (1, 2) and x.shape[1] > x.shape[0]: | |
| x = x.T | |
| # Ensure mono is (n_samples,) | |
| if x.ndim == 2 and x.shape[1] == 1: | |
| x = x[:, 0] | |
| x = x.astype(np.float32) | |
| x = np.clip(x, -1.0, 1.0) | |
| # Gradio Audio expects int16; convert here so Gradio doesn't convert and warn | |
| x = (x * 32767).astype(np.int16) | |
| return (sr, x) | |
| def spec_image( | |
| audio_array: np.ndarray, | |
| sr: int, | |
| n_fft: int = 2048, | |
| hop_length: int = 512, | |
| n_mels: int = 128, | |
| fmax: Optional[float] = None, | |
| ) -> Image.Image: | |
| """ | |
| Generate a mel-spectrogram image from an audio array. | |
| """ | |
| y = audio_array.flatten() # Ensure it's 1D | |
| S = librosa.feature.melspectrogram( | |
| y=y, | |
| sr=sr, | |
| n_fft=n_fft, | |
| hop_length=hop_length, | |
| n_mels=n_mels, | |
| fmax=fmax or sr // 2, | |
| ) | |
| S_db = librosa.power_to_db(S, ref=np.max(S)) | |
| fig, ax = plt.subplots(figsize=(8, 3), dpi=150) | |
| img = librosa.display.specshow( | |
| S_db, sr=sr, hop_length=hop_length, x_axis="time", y_axis="mel", ax=ax | |
| ) | |
| cbar = fig.colorbar(img, ax=ax, format="%+2.0f dB") | |
| cbar.set_label("dB") | |
| ax.set_title("Mel-spectrogram") | |
| ax.set_xlabel("Time in s") | |
| ax.set_ylabel("Frequency in Hz") | |
| fig.tight_layout(pad=0.2) | |
| buf = io.BytesIO() | |
| fig.savefig(buf, format="png", bbox_inches="tight", pad_inches=0) | |
| plt.close(fig) | |
| buf.seek(0) | |
| return Image.open(buf).convert("RGB") | |
| def compute_wer(reference: str, hypothesis: str) -> float: | |
| """ | |
| Compute Word Error Rate (WER) between reference and hypothesis transcripts. | |
| """ | |
| ref_words = reference.split() | |
| hyp_words = hypothesis.split() | |
| d = np.zeros((len(ref_words) + 1, len(hyp_words) + 1), dtype=np.uint8) | |
| for i in range(len(ref_words) + 1): | |
| d[i][0] = i | |
| for j in range(len(hyp_words) + 1): | |
| d[0][j] = j | |
| for i in range(1, len(ref_words) + 1): | |
| for j in range(1, len(hyp_words) + 1): | |
| if ref_words[i - 1] == hyp_words[j - 1]: | |
| cost = 0 | |
| else: | |
| cost = 1 | |
| d[i][j] = min( | |
| d[i - 1][j] + 1, # Deletion | |
| d[i][j - 1] + 1, # Insertion | |
| d[i - 1][j - 1] + cost, # Substitution | |
| ) | |
| wer = d[len(ref_words)][len(hyp_words)] / max(len(ref_words), 1) | |
| return wer | |
| def measure_loudness(x: np.ndarray, sr: int) -> float: | |
| meter = pyln.Meter(sr) | |
| return float(meter.integrated_loudness(x)) | |
| def true_peak_limiter(x: np.ndarray, sr: int, max_true_peak: float = TARGET_TP) -> np.ndarray: | |
| upsampled_sr = 192000 | |
| x_upsampled = librosa.resample(x, orig_sr=sr, target_sr=upsampled_sr) | |
| true_peak = np.max(np.abs(x_upsampled)) | |
| if true_peak > 0: | |
| true_peak_db = 20 * np.log10(true_peak) | |
| if true_peak_db > max_true_peak: | |
| gain_db = max_true_peak - true_peak_db | |
| gain = 10 ** (gain_db / 20) | |
| x_upsampled = x_upsampled * gain | |
| x_limited = librosa.resample(x_upsampled, orig_sr=upsampled_sr, target_sr=sr) | |
| x_limited = librosa.util.fix_length(x_limited, size=x.shape[-1]) | |
| return x_limited.astype("float32") | |
| def normalize_lufs(x: np.ndarray, sr: int) -> np.ndarray: | |
| """ | |
| Normalize audio to a fixed integrated loudness target and limit true peak. | |
| """ | |
| try: | |
| current_lufs = measure_loudness(x, sr) | |
| if not np.isfinite(current_lufs): | |
| return x.astype("float32") | |
| gain_db = TARGET_LOUDNESS - current_lufs | |
| gain = 10 ** (gain_db / 20) | |
| y = x * gain | |
| y = true_peak_limiter(y, sr, max_true_peak=TARGET_TP) | |
| return y.astype("float32") | |
| except Exception as e: | |
| warnings.warn(f"LUFS normalization failed, returning input unchanged: {e}") | |
| return x.astype("float32") | |