Spaces:
Running on CPU Upgrade
Running on CPU Upgrade
normalization (#2)
Browse files- Add audio normalization (4be2da0bf176acbde6dadb3dfba939e8c7a44055)
- Merge branch 'main' into pr/2 (f25152fa4bbe89c2f0edd4ff57cf4a2735473914)
- fix current_sample_rate input/output parameters (e86287a7dacd86662f995411725b92a973fcaaa2)
- app.py +19 -7
- constants.py +3 -0
- offline_pipeline.py +10 -6
- requirements.txt +2 -1
- utils.py +50 -5
app.py
CHANGED
|
@@ -71,7 +71,7 @@ def process_with_live_transcript(
|
|
| 71 |
result_holder["error"] = e
|
| 72 |
|
| 73 |
# 1) First yield: ground truth + input spectrogram only (no audio, no enhanced spec, no transcripts yet)
|
| 74 |
-
|
| 75 |
noisy_spec_path = f"{APP_TMP_DIR}/{sample_stem}_noisy_spectrogram.png"
|
| 76 |
if input_array is not None:
|
| 77 |
try:
|
|
@@ -241,10 +241,17 @@ with gr.Blocks() as demo:
|
|
| 241 |
with gr.Tab("Upload local file") as upload_tab:
|
| 242 |
with gr.Row():
|
| 243 |
gr.Markdown(open("docs/local_file.md", "r", encoding="utf-8").read())
|
| 244 |
-
audio_file_upload = gr.
|
| 245 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 246 |
)
|
| 247 |
-
|
| 248 |
enhance_btn = gr.Button("Enhance with Quail Voice Focus 2.0", scale=2, visible=False)
|
| 249 |
|
| 250 |
with gr.Group(elem_classes="panel results-card", visible=False) as results_card:
|
|
@@ -388,12 +395,17 @@ with gr.Blocks() as demo:
|
|
| 388 |
# Uploading a local file triggers loading the audio file and hiding results until enhancement
|
| 389 |
audio_file_upload.change(
|
| 390 |
lambda: gr.update(visible=False),
|
| 391 |
-
inputs=None,
|
| 392 |
outputs=results_card,
|
| 393 |
).then(
|
| 394 |
load_local_file,
|
| 395 |
-
inputs=[audio_file_upload],
|
| 396 |
-
outputs=[input_array, sample_stem, current_sample_rate]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 397 |
)
|
| 398 |
|
| 399 |
# Enhancement button: run pipeline with live transcript progress (dataset + local file modes).
|
|
|
|
| 71 |
result_holder["error"] = e
|
| 72 |
|
| 73 |
# 1) First yield: ground truth + input spectrogram only (no audio, no enhanced spec, no transcripts yet)
|
| 74 |
+
_ = cleanup_previous_run(last_sample_stem)
|
| 75 |
noisy_spec_path = f"{APP_TMP_DIR}/{sample_stem}_noisy_spectrogram.png"
|
| 76 |
if input_array is not None:
|
| 77 |
try:
|
|
|
|
| 241 |
with gr.Tab("Upload local file") as upload_tab:
|
| 242 |
with gr.Row():
|
| 243 |
gr.Markdown(open("docs/local_file.md", "r", encoding="utf-8").read())
|
| 244 |
+
audio_file_upload = gr.File(
|
| 245 |
+
file_types=[".wav", ".mp3", ".flac", ".m4a", ".ogg"],
|
| 246 |
+
file_count="single",
|
| 247 |
+
scale=3,
|
| 248 |
+
)
|
| 249 |
+
normalize = gr.Checkbox(label="Normalize audio", value=True)
|
| 250 |
+
audio_preview = gr.Audio(
|
| 251 |
+
label="Preview",
|
| 252 |
+
autoplay=False,
|
| 253 |
+
interactive=False,
|
| 254 |
)
|
|
|
|
| 255 |
enhance_btn = gr.Button("Enhance with Quail Voice Focus 2.0", scale=2, visible=False)
|
| 256 |
|
| 257 |
with gr.Group(elem_classes="panel results-card", visible=False) as results_card:
|
|
|
|
| 395 |
# Uploading a local file triggers loading the audio file and hiding results until enhancement
|
| 396 |
audio_file_upload.change(
|
| 397 |
lambda: gr.update(visible=False),
|
|
|
|
| 398 |
outputs=results_card,
|
| 399 |
).then(
|
| 400 |
load_local_file,
|
| 401 |
+
inputs=[audio_file_upload, normalize],
|
| 402 |
+
outputs=[input_array, sample_stem, audio_preview, current_sample_rate],
|
| 403 |
+
)
|
| 404 |
+
|
| 405 |
+
normalize.change(
|
| 406 |
+
load_local_file,
|
| 407 |
+
inputs=[audio_file_upload, normalize],
|
| 408 |
+
outputs=[input_array, sample_stem, audio_preview, current_sample_rate],
|
| 409 |
)
|
| 410 |
|
| 411 |
# Enhancement button: run pipeline with live transcript progress (dataset + local file modes).
|
constants.py
CHANGED
|
@@ -23,6 +23,9 @@ DEFAULT_SR: Final = 16000
|
|
| 23 |
STREAM_EVERY: Final = 0.2
|
| 24 |
WARMUP_SECONDS: Final = 2 # seconds before "recording ready" light turns on
|
| 25 |
|
|
|
|
|
|
|
|
|
|
| 26 |
STREAMER_CLASSES: Final = {
|
| 27 |
"Deepgram Nova-3 RT": DeepgramStreamer,
|
| 28 |
"Soniox STT-RT v3": SonioxStreamer,
|
|
|
|
| 23 |
STREAM_EVERY: Final = 0.2
|
| 24 |
WARMUP_SECONDS: Final = 2 # seconds before "recording ready" light turns on
|
| 25 |
|
| 26 |
+
TARGET_LOUDNESS: Final = -17.0
|
| 27 |
+
TARGET_TP: Final = -1.5
|
| 28 |
+
|
| 29 |
STREAMER_CLASSES: Final = {
|
| 30 |
"Deepgram Nova-3 RT": DeepgramStreamer,
|
| 31 |
"Soniox STT-RT v3": SonioxStreamer,
|
offline_pipeline.py
CHANGED
|
@@ -1,9 +1,10 @@
|
|
| 1 |
import os
|
|
|
|
| 2 |
|
| 3 |
import gradio as gr
|
| 4 |
import soundfile as sf
|
| 5 |
from sdk import SDKWrapper
|
| 6 |
-
from utils import spec_image, compute_wer, to_gradio_audio
|
| 7 |
from hf_dataset_utils import get_audio, get_transcript
|
| 8 |
from constants import APP_TMP_DIR, STREAMER_CLASSES
|
| 9 |
import numpy as np
|
|
@@ -123,17 +124,20 @@ def run_offline_pipeline_streaming(
|
|
| 123 |
)
|
| 124 |
|
| 125 |
def load_local_file(
|
| 126 |
-
sample_path: str
|
| 127 |
-
|
|
|
|
| 128 |
if not sample_path or not os.path.exists(sample_path):
|
| 129 |
-
|
| 130 |
-
raise ValueError("Missing audio sample. Please upload an audio sample or use the microphone input.")
|
| 131 |
if os.path.getsize(sample_path) > 5 * 1024 * 1024:
|
| 132 |
gr.Warning("File size exceeds 5 MB limit. Please upload a smaller file.")
|
| 133 |
raise ValueError("Uploaded file exceeds the 5 MB size limit.")
|
| 134 |
new_sample_stem = os.path.splitext(os.path.basename(sample_path))[0]
|
| 135 |
y, sample_rate = sf.read(sample_path, dtype="float32", always_2d=False)
|
| 136 |
-
|
|
|
|
|
|
|
|
|
|
| 137 |
|
| 138 |
def load_file_from_dataset(sample_id: str) -> tuple[tuple | None, np.ndarray | None, str, int | None]:
|
| 139 |
if not sample_id:
|
|
|
|
| 1 |
import os
|
| 2 |
+
from random import sample
|
| 3 |
|
| 4 |
import gradio as gr
|
| 5 |
import soundfile as sf
|
| 6 |
from sdk import SDKWrapper
|
| 7 |
+
from utils import spec_image, compute_wer, to_gradio_audio, normalize_lufs
|
| 8 |
from hf_dataset_utils import get_audio, get_transcript
|
| 9 |
from constants import APP_TMP_DIR, STREAMER_CLASSES
|
| 10 |
import numpy as np
|
|
|
|
| 124 |
)
|
| 125 |
|
| 126 |
def load_local_file(
|
| 127 |
+
sample_path: str,
|
| 128 |
+
normalize: bool = True,
|
| 129 |
+
) -> tuple[np.ndarray | None, str, tuple | None, int]:
|
| 130 |
if not sample_path or not os.path.exists(sample_path):
|
| 131 |
+
return None, "", None
|
|
|
|
| 132 |
if os.path.getsize(sample_path) > 5 * 1024 * 1024:
|
| 133 |
gr.Warning("File size exceeds 5 MB limit. Please upload a smaller file.")
|
| 134 |
raise ValueError("Uploaded file exceeds the 5 MB size limit.")
|
| 135 |
new_sample_stem = os.path.splitext(os.path.basename(sample_path))[0]
|
| 136 |
y, sample_rate = sf.read(sample_path, dtype="float32", always_2d=False)
|
| 137 |
+
if normalize:
|
| 138 |
+
y = normalize_lufs(y, sample_rate)
|
| 139 |
+
gradio_audio = to_gradio_audio(y, sample_rate)
|
| 140 |
+
return y, new_sample_stem, gradio_audio, sample_rate
|
| 141 |
|
| 142 |
def load_file_from_dataset(sample_id: str) -> tuple[tuple | None, np.ndarray | None, str, int | None]:
|
| 143 |
if not sample_id:
|
requirements.txt
CHANGED
|
@@ -13,4 +13,5 @@ soxr
|
|
| 13 |
datasets
|
| 14 |
torchcodec
|
| 15 |
torch
|
| 16 |
-
torchaudio
|
|
|
|
|
|
| 13 |
datasets
|
| 14 |
torchcodec
|
| 15 |
torch
|
| 16 |
+
torchaudio
|
| 17 |
+
pyloudnorm
|
utils.py
CHANGED
|
@@ -1,12 +1,13 @@
|
|
| 1 |
-
from typing import
|
| 2 |
-
|
| 3 |
import numpy as np
|
| 4 |
import librosa
|
| 5 |
from PIL import Image
|
| 6 |
import io
|
| 7 |
import matplotlib.pyplot as plt
|
| 8 |
-
import
|
| 9 |
-
|
|
|
|
|
|
|
| 10 |
|
| 11 |
def to_gradio_audio(x: np.ndarray, sr: int) -> tuple[int, np.ndarray]:
|
| 12 |
"""Return (sample_rate, int16 mono array) for Gradio Audio. Gradio expects int16;
|
|
@@ -93,4 +94,48 @@ def compute_wer(reference: str, hypothesis: str) -> float:
|
|
| 93 |
d[i - 1][j - 1] + cost, # Substitution
|
| 94 |
)
|
| 95 |
wer = d[len(ref_words)][len(hyp_words)] / max(len(ref_words), 1)
|
| 96 |
-
return wer
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from typing import Optional
|
|
|
|
| 2 |
import numpy as np
|
| 3 |
import librosa
|
| 4 |
from PIL import Image
|
| 5 |
import io
|
| 6 |
import matplotlib.pyplot as plt
|
| 7 |
+
from constants import DEFAULT_SR, TARGET_LOUDNESS, TARGET_TP
|
| 8 |
+
import warnings
|
| 9 |
+
import pyloudnorm as pyln
|
| 10 |
+
|
| 11 |
|
| 12 |
def to_gradio_audio(x: np.ndarray, sr: int) -> tuple[int, np.ndarray]:
|
| 13 |
"""Return (sample_rate, int16 mono array) for Gradio Audio. Gradio expects int16;
|
|
|
|
| 94 |
d[i - 1][j - 1] + cost, # Substitution
|
| 95 |
)
|
| 96 |
wer = d[len(ref_words)][len(hyp_words)] / max(len(ref_words), 1)
|
| 97 |
+
return wer
|
| 98 |
+
|
| 99 |
+
|
| 100 |
+
def measure_loudness(x: np.ndarray, sr: int) -> float:
|
| 101 |
+
meter = pyln.Meter(sr)
|
| 102 |
+
return float(meter.integrated_loudness(x))
|
| 103 |
+
|
| 104 |
+
|
| 105 |
+
def true_peak_limiter(x: np.ndarray, sr: int, max_true_peak: float = TARGET_TP) -> np.ndarray:
|
| 106 |
+
upsampled_sr = 192000
|
| 107 |
+
x_upsampled = librosa.resample(x, orig_sr=sr, target_sr=upsampled_sr)
|
| 108 |
+
true_peak = np.max(np.abs(x_upsampled))
|
| 109 |
+
|
| 110 |
+
if true_peak > 0:
|
| 111 |
+
true_peak_db = 20 * np.log10(true_peak)
|
| 112 |
+
if true_peak_db > max_true_peak:
|
| 113 |
+
gain_db = max_true_peak - true_peak_db
|
| 114 |
+
gain = 10 ** (gain_db / 20)
|
| 115 |
+
x_upsampled = x_upsampled * gain
|
| 116 |
+
|
| 117 |
+
x_limited = librosa.resample(x_upsampled, orig_sr=upsampled_sr, target_sr=sr)
|
| 118 |
+
x_limited = librosa.util.fix_length(x_limited, size=x.shape[-1])
|
| 119 |
+
return x_limited.astype("float32")
|
| 120 |
+
|
| 121 |
+
|
| 122 |
+
def normalize_lufs(x: np.ndarray, sr: int) -> np.ndarray:
|
| 123 |
+
"""
|
| 124 |
+
Normalize audio to a fixed integrated loudness target and limit true peak.
|
| 125 |
+
"""
|
| 126 |
+
try:
|
| 127 |
+
current_lufs = measure_loudness(x, sr)
|
| 128 |
+
|
| 129 |
+
if not np.isfinite(current_lufs):
|
| 130 |
+
return x.astype("float32")
|
| 131 |
+
|
| 132 |
+
gain_db = TARGET_LOUDNESS - current_lufs
|
| 133 |
+
gain = 10 ** (gain_db / 20)
|
| 134 |
+
|
| 135 |
+
y = x * gain
|
| 136 |
+
y = true_peak_limiter(y, sr, max_true_peak=TARGET_TP)
|
| 137 |
+
|
| 138 |
+
return y.astype("float32")
|
| 139 |
+
except Exception as e:
|
| 140 |
+
warnings.warn(f"LUFS normalization failed, returning input unchanged: {e}")
|
| 141 |
+
return x.astype("float32")
|