user4-33 commited on
Commit
909a8c1
·
1 Parent(s): c62a089

normalization (#2)

Browse files

- Add audio normalization (4be2da0bf176acbde6dadb3dfba939e8c7a44055)
- Merge branch 'main' into pr/2 (f25152fa4bbe89c2f0edd4ff57cf4a2735473914)
- fix current_sample_rate input/output parameters (e86287a7dacd86662f995411725b92a973fcaaa2)

Files changed (5) hide show
  1. app.py +19 -7
  2. constants.py +3 -0
  3. offline_pipeline.py +10 -6
  4. requirements.txt +2 -1
  5. utils.py +50 -5
app.py CHANGED
@@ -71,7 +71,7 @@ def process_with_live_transcript(
71
  result_holder["error"] = e
72
 
73
  # 1) First yield: ground truth + input spectrogram only (no audio, no enhanced spec, no transcripts yet)
74
- cleanup_out = cleanup_previous_run(last_sample_stem)
75
  noisy_spec_path = f"{APP_TMP_DIR}/{sample_stem}_noisy_spectrogram.png"
76
  if input_array is not None:
77
  try:
@@ -241,10 +241,17 @@ with gr.Blocks() as demo:
241
  with gr.Tab("Upload local file") as upload_tab:
242
  with gr.Row():
243
  gr.Markdown(open("docs/local_file.md", "r", encoding="utf-8").read())
244
- audio_file_upload = gr.Audio(
245
- type="filepath", sources=["upload"], buttons=["download"], autoplay=False, format= "wav"
 
 
 
 
 
 
 
 
246
  )
247
-
248
  enhance_btn = gr.Button("Enhance with Quail Voice Focus 2.0", scale=2, visible=False)
249
 
250
  with gr.Group(elem_classes="panel results-card", visible=False) as results_card:
@@ -388,12 +395,17 @@ with gr.Blocks() as demo:
388
  # Uploading a local file triggers loading the audio file and hiding results until enhancement
389
  audio_file_upload.change(
390
  lambda: gr.update(visible=False),
391
- inputs=None,
392
  outputs=results_card,
393
  ).then(
394
  load_local_file,
395
- inputs=[audio_file_upload],
396
- outputs=[input_array, sample_stem, current_sample_rate]
 
 
 
 
 
 
397
  )
398
 
399
  # Enhancement button: run pipeline with live transcript progress (dataset + local file modes).
 
71
  result_holder["error"] = e
72
 
73
  # 1) First yield: ground truth + input spectrogram only (no audio, no enhanced spec, no transcripts yet)
74
+ _ = cleanup_previous_run(last_sample_stem)
75
  noisy_spec_path = f"{APP_TMP_DIR}/{sample_stem}_noisy_spectrogram.png"
76
  if input_array is not None:
77
  try:
 
241
  with gr.Tab("Upload local file") as upload_tab:
242
  with gr.Row():
243
  gr.Markdown(open("docs/local_file.md", "r", encoding="utf-8").read())
244
+ audio_file_upload = gr.File(
245
+ file_types=[".wav", ".mp3", ".flac", ".m4a", ".ogg"],
246
+ file_count="single",
247
+ scale=3,
248
+ )
249
+ normalize = gr.Checkbox(label="Normalize audio", value=True)
250
+ audio_preview = gr.Audio(
251
+ label="Preview",
252
+ autoplay=False,
253
+ interactive=False,
254
  )
 
255
  enhance_btn = gr.Button("Enhance with Quail Voice Focus 2.0", scale=2, visible=False)
256
 
257
  with gr.Group(elem_classes="panel results-card", visible=False) as results_card:
 
395
  # Uploading a local file triggers loading the audio file and hiding results until enhancement
396
  audio_file_upload.change(
397
  lambda: gr.update(visible=False),
 
398
  outputs=results_card,
399
  ).then(
400
  load_local_file,
401
+ inputs=[audio_file_upload, normalize],
402
+ outputs=[input_array, sample_stem, audio_preview, current_sample_rate],
403
+ )
404
+
405
+ normalize.change(
406
+ load_local_file,
407
+ inputs=[audio_file_upload, normalize],
408
+ outputs=[input_array, sample_stem, audio_preview, current_sample_rate],
409
  )
410
 
411
  # Enhancement button: run pipeline with live transcript progress (dataset + local file modes).
constants.py CHANGED
@@ -23,6 +23,9 @@ DEFAULT_SR: Final = 16000
23
  STREAM_EVERY: Final = 0.2
24
  WARMUP_SECONDS: Final = 2 # seconds before "recording ready" light turns on
25
 
 
 
 
26
  STREAMER_CLASSES: Final = {
27
  "Deepgram Nova-3 RT": DeepgramStreamer,
28
  "Soniox STT-RT v3": SonioxStreamer,
 
23
  STREAM_EVERY: Final = 0.2
24
  WARMUP_SECONDS: Final = 2 # seconds before "recording ready" light turns on
25
 
26
+ TARGET_LOUDNESS: Final = -17.0
27
+ TARGET_TP: Final = -1.5
28
+
29
  STREAMER_CLASSES: Final = {
30
  "Deepgram Nova-3 RT": DeepgramStreamer,
31
  "Soniox STT-RT v3": SonioxStreamer,
offline_pipeline.py CHANGED
@@ -1,9 +1,10 @@
1
  import os
 
2
 
3
  import gradio as gr
4
  import soundfile as sf
5
  from sdk import SDKWrapper
6
- from utils import spec_image, compute_wer, to_gradio_audio
7
  from hf_dataset_utils import get_audio, get_transcript
8
  from constants import APP_TMP_DIR, STREAMER_CLASSES
9
  import numpy as np
@@ -123,17 +124,20 @@ def run_offline_pipeline_streaming(
123
  )
124
 
125
  def load_local_file(
126
- sample_path: str
127
- ) -> tuple[np.ndarray, str, int]:
 
128
  if not sample_path or not os.path.exists(sample_path):
129
- gr.Warning("Please upload a valid audio file.")
130
- raise ValueError("Missing audio sample. Please upload an audio sample or use the microphone input.")
131
  if os.path.getsize(sample_path) > 5 * 1024 * 1024:
132
  gr.Warning("File size exceeds 5 MB limit. Please upload a smaller file.")
133
  raise ValueError("Uploaded file exceeds the 5 MB size limit.")
134
  new_sample_stem = os.path.splitext(os.path.basename(sample_path))[0]
135
  y, sample_rate = sf.read(sample_path, dtype="float32", always_2d=False)
136
- return y, new_sample_stem, sample_rate
 
 
 
137
 
138
  def load_file_from_dataset(sample_id: str) -> tuple[tuple | None, np.ndarray | None, str, int | None]:
139
  if not sample_id:
 
1
  import os
2
+ from random import sample
3
 
4
  import gradio as gr
5
  import soundfile as sf
6
  from sdk import SDKWrapper
7
+ from utils import spec_image, compute_wer, to_gradio_audio, normalize_lufs
8
  from hf_dataset_utils import get_audio, get_transcript
9
  from constants import APP_TMP_DIR, STREAMER_CLASSES
10
  import numpy as np
 
124
  )
125
 
126
  def load_local_file(
127
+ sample_path: str,
128
+ normalize: bool = True,
129
+ ) -> tuple[np.ndarray | None, str, tuple | None, int]:
130
  if not sample_path or not os.path.exists(sample_path):
131
+ return None, "", None
 
132
  if os.path.getsize(sample_path) > 5 * 1024 * 1024:
133
  gr.Warning("File size exceeds 5 MB limit. Please upload a smaller file.")
134
  raise ValueError("Uploaded file exceeds the 5 MB size limit.")
135
  new_sample_stem = os.path.splitext(os.path.basename(sample_path))[0]
136
  y, sample_rate = sf.read(sample_path, dtype="float32", always_2d=False)
137
+ if normalize:
138
+ y = normalize_lufs(y, sample_rate)
139
+ gradio_audio = to_gradio_audio(y, sample_rate)
140
+ return y, new_sample_stem, gradio_audio, sample_rate
141
 
142
  def load_file_from_dataset(sample_id: str) -> tuple[tuple | None, np.ndarray | None, str, int | None]:
143
  if not sample_id:
requirements.txt CHANGED
@@ -13,4 +13,5 @@ soxr
13
  datasets
14
  torchcodec
15
  torch
16
- torchaudio
 
 
13
  datasets
14
  torchcodec
15
  torch
16
+ torchaudio
17
+ pyloudnorm
utils.py CHANGED
@@ -1,12 +1,13 @@
1
- from typing import Callable, Optional
2
-
3
  import numpy as np
4
  import librosa
5
  from PIL import Image
6
  import io
7
  import matplotlib.pyplot as plt
8
- import resampy
9
- from constants import DEFAULT_SR, STREAMER_CLASSES
 
 
10
 
11
  def to_gradio_audio(x: np.ndarray, sr: int) -> tuple[int, np.ndarray]:
12
  """Return (sample_rate, int16 mono array) for Gradio Audio. Gradio expects int16;
@@ -93,4 +94,48 @@ def compute_wer(reference: str, hypothesis: str) -> float:
93
  d[i - 1][j - 1] + cost, # Substitution
94
  )
95
  wer = d[len(ref_words)][len(hyp_words)] / max(len(ref_words), 1)
96
- return wer
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from typing import Optional
 
2
  import numpy as np
3
  import librosa
4
  from PIL import Image
5
  import io
6
  import matplotlib.pyplot as plt
7
+ from constants import DEFAULT_SR, TARGET_LOUDNESS, TARGET_TP
8
+ import warnings
9
+ import pyloudnorm as pyln
10
+
11
 
12
  def to_gradio_audio(x: np.ndarray, sr: int) -> tuple[int, np.ndarray]:
13
  """Return (sample_rate, int16 mono array) for Gradio Audio. Gradio expects int16;
 
94
  d[i - 1][j - 1] + cost, # Substitution
95
  )
96
  wer = d[len(ref_words)][len(hyp_words)] / max(len(ref_words), 1)
97
+ return wer
98
+
99
+
100
+ def measure_loudness(x: np.ndarray, sr: int) -> float:
101
+ meter = pyln.Meter(sr)
102
+ return float(meter.integrated_loudness(x))
103
+
104
+
105
+ def true_peak_limiter(x: np.ndarray, sr: int, max_true_peak: float = TARGET_TP) -> np.ndarray:
106
+ upsampled_sr = 192000
107
+ x_upsampled = librosa.resample(x, orig_sr=sr, target_sr=upsampled_sr)
108
+ true_peak = np.max(np.abs(x_upsampled))
109
+
110
+ if true_peak > 0:
111
+ true_peak_db = 20 * np.log10(true_peak)
112
+ if true_peak_db > max_true_peak:
113
+ gain_db = max_true_peak - true_peak_db
114
+ gain = 10 ** (gain_db / 20)
115
+ x_upsampled = x_upsampled * gain
116
+
117
+ x_limited = librosa.resample(x_upsampled, orig_sr=upsampled_sr, target_sr=sr)
118
+ x_limited = librosa.util.fix_length(x_limited, size=x.shape[-1])
119
+ return x_limited.astype("float32")
120
+
121
+
122
+ def normalize_lufs(x: np.ndarray, sr: int) -> np.ndarray:
123
+ """
124
+ Normalize audio to a fixed integrated loudness target and limit true peak.
125
+ """
126
+ try:
127
+ current_lufs = measure_loudness(x, sr)
128
+
129
+ if not np.isfinite(current_lufs):
130
+ return x.astype("float32")
131
+
132
+ gain_db = TARGET_LOUDNESS - current_lufs
133
+ gain = 10 ** (gain_db / 20)
134
+
135
+ y = x * gain
136
+ y = true_peak_limiter(y, sr, max_true_peak=TARGET_TP)
137
+
138
+ return y.astype("float32")
139
+ except Exception as e:
140
+ warnings.warn(f"LUFS normalization failed, returning input unchanged: {e}")
141
+ return x.astype("float32")