jj-wohlgemuth commited on
Commit
4f24eb5
·
1 Parent(s): db67085

extended intro and limited file size to 5MB

Browse files
Files changed (2) hide show
  1. docs/intro.md +13 -2
  2. offline_pipeline.py +9 -2
docs/intro.md CHANGED
@@ -1,3 +1,14 @@
1
- Welcome! Try **ai‑coustics Quail Voice Focus**. This is our real‑time enhancement model designed for speech-to-text, which **isolates the main speaker** and reduces background voice acitivity. Learn more in our [documentation](https://docs.ai-coustics.com/guides/models#quail).
2
 
3
- This model is optimized to keep the phonetic details that speech-to-text systems need, so the output may not always sound "prettier," but delivers the most accurate transcriptions.
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Welcome! Try ai‑coustics Quail Voice Focus 2.0
2
 
3
+ Our flagship real‑time enhancement model is engineered specifically for Speech-to-Text (STT). By isolating the foreground speaker and suppressing background chatter, it preserves the critical phonetic details that AI systems need to transcribe accurately.
4
+
5
+ > **Note:** This model is optimized for machine performance rather than perceptual quality. The output may sound less processed to a listener, but this ensures higher transcription accuracy.
6
+
7
+ ## 🚀 Now Native in LiveKit & Pipecat
8
+
9
+ Build production-ready voice agents with ai‑coustics audio intelligence available natively in the most widely used real-time infrastructure layers:
10
+
11
+ - **[LiveKit](https://github.com/livekit/plugins-ai-coustics-python)** — official ai‑coustics plugin for LiveKit
12
+ - **[Pipecat](https://github.com/pipecat-ai/pipecat/blob/main/src/pipecat/audio/filters/aic_filter.py)** — ai‑coustics audio filter for Pipecat
13
+
14
+ [Learn more in our documentation](https://docs.ai-coustics.com/guides/models#quail-voice-focus-2-0-l-16-khz)
offline_pipeline.py CHANGED
@@ -14,7 +14,8 @@ def retrieve_audio_information(
14
  sample_id: str,
15
  stt_model: str,
16
  ) -> tuple[str, str, str, str, str, str]:
17
-
 
18
  noisy_spec_path = f"/tmp/{sample_id}_noisy_spectrogram.png"
19
  enhanced_spec_path = f"/tmp/{sample_id}_enhanced_spectrogram.png"
20
  spec_image(original_array).save(noisy_spec_path)
@@ -33,9 +34,12 @@ def retrieve_audio_information(
33
 
34
 
35
  def denoise_audio(
36
- sample_16k: np.ndarray,
37
  enhancement_level: float = 50.0,
38
  ) -> tuple[np.ndarray | None , tuple[int, np.ndarray]| None]:
 
 
 
39
  try:
40
  sdk = SDKWrapper()
41
  sdk.init_processor(sample_rate=DEFAULT_SR, enhancement_level=float(enhancement_level) / 100.0)
@@ -53,6 +57,9 @@ def load_local_file(
53
  if not sample_path or not os.path.exists(sample_path):
54
  gr.Warning("Please upload a valid audio file.")
55
  raise ValueError("Missing audio sample. Please upload an audio sample or use the microphone input.")
 
 
 
56
  new_sample_stem = os.path.splitext(os.path.basename(sample_path))[0]
57
  y_16k, _ = librosa.load(sample_path, sr=DEFAULT_SR, dtype="float32", mono=True)
58
  return y_16k, new_sample_stem
 
14
  sample_id: str,
15
  stt_model: str,
16
  ) -> tuple[str, str, str, str, str, str]:
17
+ if original_array is None or enhanced_array is None:
18
+ raise ValueError("Audio arrays are not available.")
19
  noisy_spec_path = f"/tmp/{sample_id}_noisy_spectrogram.png"
20
  enhanced_spec_path = f"/tmp/{sample_id}_enhanced_spectrogram.png"
21
  spec_image(original_array).save(noisy_spec_path)
 
34
 
35
 
36
  def denoise_audio(
37
+ sample_16k: np.ndarray,
38
  enhancement_level: float = 50.0,
39
  ) -> tuple[np.ndarray | None , tuple[int, np.ndarray]| None]:
40
+ if sample_16k is None:
41
+ gr.Warning("No audio loaded. Please upload a file first.")
42
+ raise ValueError("No audio to enhance.")
43
  try:
44
  sdk = SDKWrapper()
45
  sdk.init_processor(sample_rate=DEFAULT_SR, enhancement_level=float(enhancement_level) / 100.0)
 
57
  if not sample_path or not os.path.exists(sample_path):
58
  gr.Warning("Please upload a valid audio file.")
59
  raise ValueError("Missing audio sample. Please upload an audio sample or use the microphone input.")
60
+ if os.path.getsize(sample_path) > 5 * 1024 * 1024:
61
+ gr.Warning("File size exceeds 5 MB limit. Please upload a smaller file.")
62
+ raise ValueError("Uploaded file exceeds the 5 MB size limit.")
63
  new_sample_stem = os.path.splitext(os.path.basename(sample_path))[0]
64
  y_16k, _ = librosa.load(sample_path, sr=DEFAULT_SR, dtype="float32", mono=True)
65
  return y_16k, new_sample_stem