mariesig commited on
Commit
f5f5219
·
1 Parent(s): debd261

extra offline file

Browse files
Files changed (2) hide show
  1. app.py +97 -290
  2. offline.py +112 -0
app.py CHANGED
@@ -1,25 +1,18 @@
1
  import os
2
  import time
3
- from typing import Optional, Any
4
 
5
  import gradio as gr
6
  from loguru import logger
7
- from online import transcribe, clear_ui, change_stt_model
8
- from constants import (
9
- MINUTES_KEEP,
10
- )
11
- from sdk import SDKWrapper
12
- from audio_tools import spec_image
13
- import shutil
14
- import tempfile
15
- from aic_dataset import ALL_FILES, get_local_mix_path, download_transcript
16
- from transcribe import transcribe_and_evaluate, transcribe_file
17
 
18
  # ===============================
19
  # Temporary File & Cache Management
20
  # ===============================
21
-
22
-
23
  def cleanup_tmp(minutes_keep: int = MINUTES_KEEP, filter: list[str] = []):
24
  skipped = 0
25
  removed = 0
@@ -29,10 +22,7 @@ def cleanup_tmp(minutes_keep: int = MINUTES_KEEP, filter: list[str] = []):
29
  f = os.path.join(root, name)
30
  is_old = (time.time() - os.path.getmtime(f)) / 60 > minutes_keep
31
  filtered = any(filt in f for filt in filter)
32
- if filtered:
33
- skipped += 1
34
- continue
35
- if not is_old:
36
  skipped += 1
37
  continue
38
  try:
@@ -43,321 +33,138 @@ def cleanup_tmp(minutes_keep: int = MINUTES_KEEP, filter: list[str] = []):
43
  logger.info(f"Cleanup tmp complete. Removed {removed} files, skipped {skipped} files.")
44
 
45
 
46
- # ===============================
47
- # Interface Logic
48
- # ===============================
49
-
50
- def transcribe_with_original(audio_file_path: str, file_stem: str, streamer_type: str = "deepgram") -> tuple[str, Any, Any]:
51
- """
52
- Transcribe an audio file and compute WER against a reference transcript.
53
-
54
- Args:
55
- audio_file_path (str): Path to WAV file
56
- file_stem (str): Base filename for the audio file
57
- streamer_type (str): "soniox" or "deepgram"
58
- Returns:
59
- tuple[str, float, Any]: Transcript text, Word Error Rate (WER), and visibility update for the original transcript
60
- """
61
- original_transcript = download_transcript(file_stem)
62
- transcript, wer = transcribe_and_evaluate(audio_file_path, original_transcript, streamer_type)
63
- wer_box = gr.update(value=wer, visible=True)
64
- original_transcript_update = gr.update(value=original_transcript, visible=True)
65
- return transcript, wer_box, original_transcript_update
66
-
67
- def transcribe_no_original(audio_file_path: str, stt_model: str = "deepgram") -> tuple[str, Any, Any]:
68
- """
69
- Transcribe an audio file without a reference transcript.
70
-
71
- Args:
72
- audio_file_path (str): Path to WAV file
73
- stt_model (str): "soniox" or "deepgram"
74
- Returns:
75
- tuple[str, float, Any]: Transcript text, Word Error Rate (WER), and visibility update for the original transcript
76
- """
77
- transcript = transcribe_file(audio_file_path, stt_model)
78
- visibility = gr.update(visible=False)
79
- return transcript, visibility, visibility
80
-
81
-
82
- def create_results_title(path: str) -> str:
83
- file_name = os.path.basename(path)
84
- if not file_name:
85
- return "## Results"
86
- return f"## Results for {file_name}"
87
-
88
-
89
- def denoise_audio(
90
- sample_path: str,
91
- enhancement_level: float = 50.0,
92
- ) -> tuple[Optional[str], Optional[str], Optional[str]]:
93
- gr.Info(
94
- "Processing started. This may take a moment. Please do not refresh or close the window."
95
- )
96
- base, ext = os.path.splitext(sample_path)
97
- enhanced_path = f"{base}_enhanced{ext}"
98
- noisy_path = f"{base}_noisy{ext}"
99
- noisy_spec_path = f"{base}_noisy_spectrogram.png"
100
- enhanced_spec_path = f"{base}_enhanced_spectrogram.png"
101
- noisy_path = sample_path
102
- try:
103
- sdk = SDKWrapper(os.getenv("SECRET_SDK_KEY"))
104
- sdk.init_processor(sample_rate=16000, enhancement_level=enhancement_level / 100)
105
- sdk.process_file(noisy_path, enhanced_path)
106
- except Exception as e:
107
- gr.Warning(f"{e}")
108
- delete_related_files(sample_path)
109
- return None, None, None
110
- noisy_im = spec_image(noisy_path)
111
- noisy_im.save(noisy_spec_path)
112
- enhanced_im = spec_image(enhanced_path)
113
- enhanced_im.save(enhanced_spec_path)
114
- print(f"Enhancement complete. id: {base}")
115
- return enhanced_path, enhanced_spec_path, noisy_spec_path
116
-
117
-
118
- def toggle_SNR(choice: str):
119
- if choice == "None":
120
- return gr.update(visible=False, value="None")
121
- else:
122
- return gr.update(visible=True, value="10")
123
-
124
-
125
- def delete_related_files(path: str):
126
- """
127
- Deletes all files in /tmp containing the base filename of the given path.
128
- """
129
- filename_no_ext = os.path.splitext(os.path.basename(path))[0]
130
- base_dir = "/tmp"
131
- deleted = 0
132
- for root, _, files in os.walk(base_dir):
133
- for f in files:
134
- if filename_no_ext in f:
135
- full_path = os.path.join(root, f)
136
- try:
137
- os.remove(full_path)
138
- deleted += 1
139
- except Exception as e:
140
- logger.warning(f"Failed to delete file {full_path}: {e}")
141
- if deleted == 0:
142
- logger.info(f"No files found to delete containing '{filename_no_ext}' in {base_dir}")
143
- logger.info(f"Deleted {deleted} files related to the last enhancement '{filename_no_ext}'.")
144
-
145
-
146
- def cleanup(last_enhancement: str, last_audio_file: str = "", new_audio_file: str = ""):
147
- """
148
- Deletes
149
- - all enhancement files (usually .png & .wav) in the /tmp directory from the last enhancement
150
- - the last uploaded audio file if a new one is uploaded
151
- - other files in /tmp older than 2 hours to help manage disk space.
152
-
153
- Note:
154
- Consider adding a flag to files to indicate whether files can be deleted, if disk usage becomes an issue.
155
- """
156
- # Delete the last uploaded audio file if a new one is uploaded
157
- if last_audio_file and last_audio_file != new_audio_file and os.path.exists(last_audio_file):
158
- try:
159
- os.remove(last_audio_file)
160
- logger.info(f"Deleted last uploaded audio file: {last_audio_file}")
161
- except Exception as e:
162
- logger.warning(f"Failed to delete last uploaded audio file {last_audio_file}: {e}")
163
- if last_enhancement:
164
- delete_related_files(last_enhancement)
165
- cleanup_tmp(minutes_keep=120)
166
-
167
-
168
- def start_processing(sample_path: str) -> tuple[str, str, Any, str]:
169
- success = True
170
- result_title = create_results_title(sample_path)
171
- if not sample_path or not os.path.exists(sample_path):
172
- gr.Warning("Please upload an audio sample or use the microphone input.")
173
- success = False
174
- if not os.getenv("SECRET_SDK_KEY"):
175
- gr.Warning("No SDK key provided. Please contact us at https://ai-coustics.com/contact/.")
176
- success = False
177
- if not success:
178
- raise ValueError("Missing audio sample or API/SDK key. Processing cannot start.")
179
- # Generate a new /tmp path with a unique filename using tempfile
180
- ext = os.path.splitext(sample_path)[1]
181
- with tempfile.NamedTemporaryFile(delete=False, suffix=ext, dir="/tmp") as tmp_file:
182
- shutil.copy(sample_path, tmp_file.name)
183
- input_enhancement_path = tmp_file.name
184
- return (
185
- input_enhancement_path,
186
- sample_path,
187
- gr.update(visible=False),
188
- result_title,
189
- )
190
-
191
-
192
-
193
  # ===============================
194
  # Gradio UI Layout
195
  # ===============================
196
  with gr.Blocks() as demo:
197
  input_enhancement = gr.State()
198
  last_audio_file = gr.State()
199
-
200
  gr.HTML(
201
- '<a href="https://ai-coustics.com/" target="_blank">'
202
- '<img src="https://mintcdn.com/ai-coustics/Sxcrv8jVSE2qWMR1/logo/dark.svg?fit=max&auto=format&n=Sxcrv8jVSE2qWMR1&q=85&s=7f26caaf21e963912961cbd8541e6d84" alt="ai-coustics Logo" width="400" style="display: block; margin: 0 auto;">'
203
- "</a>"
204
- )
 
205
  gr.Markdown(open("docs/intro.md", "r", encoding="utf-8").read())
 
 
206
  stt_model = gr.Radio(label="STT Model", choices=["Deepgram", "Soniox"], value="Deepgram", interactive=True)
207
  enhancement_level = gr.Slider(
208
- minimum=0,
209
- maximum=100,
210
- step=1,
211
- value=100,
212
- label="Enhancement level (%)",
213
- scale=2,
214
  )
215
- # ===== Tabs =====
 
 
 
216
  with gr.Tabs(elem_classes="main-tabs"):
217
  # =========================
218
- # OFFLINE TAB (FULL APP)
219
  # =========================
220
  with gr.Tab("Offline", elem_classes="tab-offline"):
221
- # ---- Input ----
222
  with gr.Group(elem_classes="panel"):
223
  with gr.Tab("Upload", elem_classes="upload-tab"):
224
- audio_file_upload = gr.Audio(
225
- type="filepath",
226
- sources=["upload", "microphone"],
227
-
228
- )
229
-
230
  enhance_btn_for_upload = gr.Button("Enhance", scale=2)
 
231
  with gr.Tab("AIC Dataset", elem_classes="dataset-tab"):
232
- dataset_dropdown = gr.Dropdown(
233
- choices=ALL_FILES,
234
- label="Choose sample",
235
- value=None,
236
- )
237
- audio_file_from_dataset = gr.Audio(
238
- type="filepath",
239
- interactive=False,
240
- )
241
-
242
  enhance_btn_for_dataset = gr.Button("Enhance", scale=2)
243
 
244
  with gr.Group(elem_classes="panel results-card", visible=False) as results_card:
245
  result_title = gr.Markdown("")
 
246
 
247
- # Full-width audio at top of the card
248
- enhanced_audio = gr.Audio(
249
- type="filepath",
250
- interactive=False
251
- )
252
-
253
- # Two columns inside one big card
254
  with gr.Row(equal_height=True, elem_classes="results-row"):
255
- # LEFT: Spectrograms
256
  with gr.Column(scale=5, min_width=320, elem_classes="results-left"):
257
- noisy_image = gr.Image(
258
- label="Input spectrogram",
259
- format="png",
260
- type="filepath",
261
- )
262
- enhanced_image = gr.Image(
263
- label="Enhanced spectrogram",
264
- format="png",
265
- type="filepath",
266
- )
267
-
268
- # RIGHT: Text + WER
269
  with gr.Column(scale=5, min_width=320, elem_classes="results-right"):
270
- original_transcript = gr.Textbox(
271
- label="Original transcript",
272
- lines=3,
273
- interactive=False,
274
- )
275
- enhanced_transcript = gr.Textbox(
276
- label="Enhanced transcript",
277
- lines=3,
278
- interactive=False,
279
- )
280
- wer_box = gr.Number(
281
- label="Word Error Rate (WER)",
282
- interactive=False,
283
- )
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
284
 
285
  # =========================
286
- # ONLINE TAB (PLACEHOLDER)
287
  # =========================
288
  with gr.Tab("Online", elem_classes="tab-online"):
289
  with gr.Group(elem_classes="panel"):
290
  stream_state = gr.State(None)
291
- audio = gr.Audio(sources=["microphone"], streaming=True)
 
 
 
 
 
292
  clear_btn = gr.Button("Clear")
293
- enhanced_text = gr.Textbox(label="Enhanced Transcribed Text", lines=6)
294
- raw_text = gr.Textbox(label="Raw Transcribed Text", lines=6)
295
- audio.stream(
296
- fn=transcribe,
297
- inputs=[stream_state, audio, enhancement_level],
298
  outputs=[stream_state, enhanced_text, raw_text],
299
  stream_every=0.05,
300
  )
301
 
302
  clear_btn.click(
303
- fn=clear_ui,
304
  outputs=[stream_state, enhanced_text, raw_text],
305
  )
306
- stt_model.change(
307
- fn=change_stt_model,
308
- inputs=stt_model,
309
- )
310
- # ===== Events / Wiring =====
311
- dataset_dropdown.change(get_local_mix_path, inputs=dataset_dropdown, outputs=[audio_file_from_dataset])
312
 
313
- #dataset_dropdown.change(create_results_title, inputs=dataset_dropdown, outputs=result_title)
314
 
315
-
316
- enhance_btn_for_dataset.click(
317
- cleanup,
318
- inputs=[input_enhancement, last_audio_file, audio_file_from_dataset],
319
- outputs=None,
320
- ).then(
321
- start_processing,
322
- inputs=audio_file_from_dataset,
323
- outputs=[input_enhancement, last_audio_file, results_card,result_title],
324
- ).success(
325
- denoise_audio,
326
- inputs=[input_enhancement, enhancement_level],
327
- outputs=[enhanced_audio, enhanced_image, noisy_image],
328
- ).then(transcribe_with_original,
329
- inputs=[enhanced_audio, dataset_dropdown, stt_model],
330
- outputs=[enhanced_transcript, wer_box, original_transcript]
331
- ).then(
332
- lambda: gr.update(visible=True),
333
- inputs=None,
334
- outputs=results_card,
335
- )
336
-
337
-
338
- enhance_btn_for_upload.click(
339
- cleanup,
340
- inputs=[input_enhancement, last_audio_file, audio_file_upload],
341
- outputs=None,
342
- ).then(
343
- start_processing,
344
- inputs=audio_file_upload,
345
- outputs=[input_enhancement, last_audio_file, results_card,result_title],
346
- ).success(
347
- denoise_audio,
348
- inputs=[input_enhancement, enhancement_level],
349
- outputs=[enhanced_audio, enhanced_image, noisy_image],
350
- ).then(
351
- transcribe_no_original,
352
- inputs=[enhanced_audio, stt_model],
353
- outputs=[enhanced_transcript, wer_box, original_transcript]
354
- ).then(
355
- lambda: gr.update(visible=True),
356
- inputs=None,
357
- outputs=results_card,
358
- )
359
-
360
-
361
-
362
  cleanup_tmp(minutes_keep=0, filter=[])
363
  demo.launch(allowed_paths=["/tmp", "/"])
 
1
  import os
2
  import time
 
3
 
4
  import gradio as gr
5
  from loguru import logger
6
+
7
+ from constants import MINUTES_KEEP
8
+ from aic_dataset import ALL_FILES, get_local_mix_path
9
+
10
+ from online import transcribe as online_transcribe, clear_ui as online_clear_ui, change_stt_model
11
+ from offline import transcribe_with_original, transcribe_no_original, denoise_audio, cleanup, start_processing
 
 
 
 
12
 
13
  # ===============================
14
  # Temporary File & Cache Management
15
  # ===============================
 
 
16
  def cleanup_tmp(minutes_keep: int = MINUTES_KEEP, filter: list[str] = []):
17
  skipped = 0
18
  removed = 0
 
22
  f = os.path.join(root, name)
23
  is_old = (time.time() - os.path.getmtime(f)) / 60 > minutes_keep
24
  filtered = any(filt in f for filt in filter)
25
+ if filtered or not is_old:
 
 
 
26
  skipped += 1
27
  continue
28
  try:
 
33
  logger.info(f"Cleanup tmp complete. Removed {removed} files, skipped {skipped} files.")
34
 
35
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
36
  # ===============================
37
  # Gradio UI Layout
38
  # ===============================
39
  with gr.Blocks() as demo:
40
  input_enhancement = gr.State()
41
  last_audio_file = gr.State()
42
+
43
  gr.HTML(
44
+ '<a href="https://ai-coustics.com/" target="_blank">'
45
+ '<img src="https://mintcdn.com/ai-coustics/Sxcrv8jVSE2qWMR1/logo/dark.svg?fit=max&auto=format&n=Sxcrv8jVSE2qWMR1&q=85&s=7f26caaf21e963912961cbd8541e6d84" '
46
+ 'alt="ai-coustics Logo" width="400" style="display: block; margin: 0 auto;">'
47
+ "</a>"
48
+ )
49
  gr.Markdown(open("docs/intro.md", "r", encoding="utf-8").read())
50
+
51
+ # ✅ Global controls (shared by both tabs)
52
  stt_model = gr.Radio(label="STT Model", choices=["Deepgram", "Soniox"], value="Deepgram", interactive=True)
53
  enhancement_level = gr.Slider(
54
+ minimum=0,
55
+ maximum=100,
56
+ step=1,
57
+ value=100,
58
+ label="Enhancement level (%)",
59
+ scale=2,
60
  )
61
+
62
+ # Online STT streamer swap uses the same global control
63
+ stt_model.change(fn=change_stt_model, inputs=stt_model, outputs=[])
64
+
65
  with gr.Tabs(elem_classes="main-tabs"):
66
  # =========================
67
+ # OFFLINE TAB
68
  # =========================
69
  with gr.Tab("Offline", elem_classes="tab-offline"):
 
70
  with gr.Group(elem_classes="panel"):
71
  with gr.Tab("Upload", elem_classes="upload-tab"):
72
+ audio_file_upload = gr.Audio(type="filepath", sources=["upload", "microphone"])
 
 
 
 
 
73
  enhance_btn_for_upload = gr.Button("Enhance", scale=2)
74
+
75
  with gr.Tab("AIC Dataset", elem_classes="dataset-tab"):
76
+ dataset_dropdown = gr.Dropdown(choices=ALL_FILES, label="Choose sample", value=None)
77
+ audio_file_from_dataset = gr.Audio(type="filepath", interactive=False)
 
 
 
 
 
 
 
 
78
  enhance_btn_for_dataset = gr.Button("Enhance", scale=2)
79
 
80
  with gr.Group(elem_classes="panel results-card", visible=False) as results_card:
81
  result_title = gr.Markdown("")
82
+ enhanced_audio = gr.Audio(type="filepath", interactive=False)
83
 
 
 
 
 
 
 
 
84
  with gr.Row(equal_height=True, elem_classes="results-row"):
 
85
  with gr.Column(scale=5, min_width=320, elem_classes="results-left"):
86
+ noisy_image = gr.Image(label="Input spectrogram", format="png", type="filepath")
87
+ enhanced_image = gr.Image(label="Enhanced spectrogram", format="png", type="filepath")
88
+
 
 
 
 
 
 
 
 
 
89
  with gr.Column(scale=5, min_width=320, elem_classes="results-right"):
90
+ original_transcript = gr.Textbox(label="Original transcript", lines=3, interactive=False)
91
+ enhanced_transcript = gr.Textbox(label="Enhanced transcript", lines=3, interactive=False)
92
+ wer_box = gr.Number(label="Word Error Rate (WER)", interactive=False)
93
+
94
+ # Wiring (offline)
95
+ dataset_dropdown.change(get_local_mix_path, inputs=dataset_dropdown, outputs=[audio_file_from_dataset])
96
+
97
+ enhance_btn_for_dataset.click(
98
+ cleanup,
99
+ inputs=[input_enhancement, last_audio_file, audio_file_from_dataset],
100
+ outputs=None,
101
+ ).then(
102
+ start_processing,
103
+ inputs=audio_file_from_dataset,
104
+ outputs=[input_enhancement, last_audio_file, results_card, result_title],
105
+ ).success(
106
+ denoise_audio,
107
+ inputs=[input_enhancement, enhancement_level],
108
+ outputs=[enhanced_audio, enhanced_image, noisy_image],
109
+ ).then(
110
+ transcribe_with_original,
111
+ inputs=[enhanced_audio, dataset_dropdown, stt_model],
112
+ outputs=[enhanced_transcript, wer_box, original_transcript],
113
+ ).then(
114
+ lambda: gr.update(visible=True),
115
+ inputs=None,
116
+ outputs=results_card,
117
+ )
118
+
119
+ enhance_btn_for_upload.click(
120
+ cleanup,
121
+ inputs=[input_enhancement, last_audio_file, audio_file_upload],
122
+ outputs=None,
123
+ ).then(
124
+ start_processing,
125
+ inputs=audio_file_upload,
126
+ outputs=[input_enhancement, last_audio_file, results_card, result_title],
127
+ ).success(
128
+ denoise_audio,
129
+ inputs=[input_enhancement, enhancement_level],
130
+ outputs=[enhanced_audio, enhanced_image, noisy_image],
131
+ ).then(
132
+ transcribe_no_original,
133
+ inputs=[enhanced_audio, stt_model],
134
+ outputs=[enhanced_transcript, wer_box, original_transcript],
135
+ ).then(
136
+ lambda: gr.update(visible=True),
137
+ inputs=None,
138
+ outputs=results_card,
139
+ )
140
 
141
  # =========================
142
+ # ONLINE TAB
143
  # =========================
144
  with gr.Tab("Online", elem_classes="tab-online"):
145
  with gr.Group(elem_classes="panel"):
146
  stream_state = gr.State(None)
147
+ audio_stream = gr.Audio(sources=["microphone"], streaming=True)
148
+ with gr.Group(elem_classes="panel"):
149
+ with gr.Column(scale=5, min_width=320):
150
+ enhanced_text = gr.Textbox(label="Enhanced Transcribed Text", lines=6)
151
+ with gr.Column(scale=5, min_width=320):
152
+ raw_text = gr.Textbox(label="Raw Transcribed Text", lines=6)
153
  clear_btn = gr.Button("Clear")
154
+
155
+
156
+ audio_stream.stream(
157
+ fn=online_transcribe,
158
+ inputs=[stream_state, audio_stream, enhancement_level],
159
  outputs=[stream_state, enhanced_text, raw_text],
160
  stream_every=0.05,
161
  )
162
 
163
  clear_btn.click(
164
+ fn=online_clear_ui,
165
  outputs=[stream_state, enhanced_text, raw_text],
166
  )
 
 
 
 
 
 
167
 
 
168
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
169
  cleanup_tmp(minutes_keep=0, filter=[])
170
  demo.launch(allowed_paths=["/tmp", "/"])
offline.py ADDED
@@ -0,0 +1,112 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import os
2
+ from typing import Optional, Any
3
+
4
+ import gradio as gr
5
+ from loguru import logger
6
+
7
+ from sdk import SDKWrapper
8
+ from audio_tools import spec_image
9
+ import shutil
10
+ import tempfile
11
+ from aic_dataset import download_transcript
12
+ from transcribe import transcribe_and_evaluate, transcribe_file
13
+
14
+
15
+ def transcribe_with_original(
16
+ audio_file_path: str,
17
+ file_stem: str,
18
+ streamer_type: str = "deepgram",
19
+ ) -> tuple[str, Any, Any]:
20
+ original_transcript = download_transcript(file_stem)
21
+ transcript, wer = transcribe_and_evaluate(audio_file_path, original_transcript, streamer_type)
22
+ wer_box = gr.update(value=wer, visible=True)
23
+ original_transcript_update = gr.update(value=original_transcript, visible=True)
24
+ return transcript, wer_box, original_transcript_update
25
+
26
+
27
+ def transcribe_no_original(audio_file_path: str, stt_model: str = "deepgram") -> tuple[str, Any, Any]:
28
+ transcript = transcribe_file(audio_file_path, stt_model)
29
+ hidden = gr.update(visible=False)
30
+ return transcript, hidden, hidden
31
+
32
+
33
+ def create_results_title(path: str) -> str:
34
+ file_name = os.path.basename(path)
35
+ return f"## Results for {file_name}" if file_name else "## Results"
36
+
37
+
38
+ # ===============================
39
+ # Enhancement (offline)
40
+ # ===============================
41
+ def denoise_audio(
42
+ sample_path: str,
43
+ enhancement_level: float = 50.0,
44
+ ) -> tuple[Optional[str], Optional[str], Optional[str]]:
45
+ gr.Info("Processing started. This may take a moment. Please do not refresh or close the window.")
46
+
47
+ base, ext = os.path.splitext(sample_path)
48
+ enhanced_path = f"{base}_enhanced{ext}"
49
+ noisy_spec_path = f"{base}_noisy_spectrogram.png"
50
+ enhanced_spec_path = f"{base}_enhanced_spectrogram.png"
51
+
52
+ try:
53
+ sdk = SDKWrapper(os.getenv("SECRET_SDK_KEY"))
54
+ sdk.init_processor(sample_rate=16000, enhancement_level=float(enhancement_level) / 100.0)
55
+ sdk.process_file(sample_path, enhanced_path)
56
+ except Exception as e:
57
+ gr.Warning(f"{e}")
58
+ delete_related_files(sample_path)
59
+ return None, None, None
60
+
61
+ spec_image(sample_path).save(noisy_spec_path)
62
+ spec_image(enhanced_path).save(enhanced_spec_path)
63
+
64
+ return enhanced_path, enhanced_spec_path, noisy_spec_path
65
+
66
+
67
+ def delete_related_files(path: str):
68
+ filename_no_ext = os.path.splitext(os.path.basename(path))[0]
69
+ base_dir = "/tmp"
70
+ deleted = 0
71
+ for root, _, files in os.walk(base_dir):
72
+ for f in files:
73
+ if filename_no_ext in f:
74
+ full_path = os.path.join(root, f)
75
+ try:
76
+ os.remove(full_path)
77
+ deleted += 1
78
+ except Exception as e:
79
+ logger.warning(f"Failed to delete file {full_path}: {e}")
80
+ logger.info(f"Deleted {deleted} files related to '{filename_no_ext}'.")
81
+
82
+
83
+ def cleanup(last_enhancement: str, last_audio_file: str = "", new_audio_file: str = ""):
84
+ # delete last uploaded audio if a new one is uploaded
85
+ if last_audio_file and last_audio_file != new_audio_file and os.path.exists(last_audio_file):
86
+ try:
87
+ os.remove(last_audio_file)
88
+ logger.info(f"Deleted last uploaded audio file: {last_audio_file}")
89
+ except Exception as e:
90
+ logger.warning(f"Failed to delete last uploaded audio file {last_audio_file}: {e}")
91
+
92
+ if last_enhancement:
93
+ delete_related_files(last_enhancement)
94
+
95
+ cleanup_tmp(minutes_keep=120)
96
+
97
+
98
+ def start_processing(sample_path: str) -> tuple[str, str, Any, str]:
99
+ if not sample_path or not os.path.exists(sample_path):
100
+ raise ValueError("Missing audio sample. Please upload an audio sample or use the microphone input.")
101
+
102
+ if not os.getenv("SECRET_SDK_KEY"):
103
+ raise ValueError("No SDK key provided. Please contact us at https://ai-coustics.com/contact/.")
104
+
105
+ result_title = create_results_title(sample_path)
106
+
107
+ ext = os.path.splitext(sample_path)[1]
108
+ with tempfile.NamedTemporaryFile(delete=False, suffix=ext, dir="/tmp") as tmp_file:
109
+ shutil.copy(sample_path, tmp_file.name)
110
+ input_enhancement_path = tmp_file.name
111
+
112
+ return input_enhancement_path, sample_path, gr.update(visible=False), result_title