Spaces:
Running on CPU Upgrade
Running on CPU Upgrade
jj-wohlgemuth commited on
Commit ·
4f24eb5
1
Parent(s): db67085
extended intro and limited file size to 5MB
Browse files- docs/intro.md +13 -2
- offline_pipeline.py +9 -2
docs/intro.md
CHANGED
|
@@ -1,3 +1,14 @@
|
|
| 1 |
-
Welcome! Try
|
| 2 |
|
| 3 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Welcome! Try ai‑coustics Quail Voice Focus 2.0
|
| 2 |
|
| 3 |
+
Our flagship real‑time enhancement model is engineered specifically for Speech-to-Text (STT). By isolating the foreground speaker and suppressing background chatter, it preserves the critical phonetic details that AI systems need to transcribe accurately.
|
| 4 |
+
|
| 5 |
+
> **Note:** This model is optimized for machine performance rather than perceptual quality. The output may sound less processed to a listener, but this ensures higher transcription accuracy.
|
| 6 |
+
|
| 7 |
+
## 🚀 Now Native in LiveKit & Pipecat
|
| 8 |
+
|
| 9 |
+
Build production-ready voice agents with ai‑coustics audio intelligence available natively in the most widely used real-time infrastructure layers:
|
| 10 |
+
|
| 11 |
+
- **[LiveKit](https://github.com/livekit/plugins-ai-coustics-python)** — official ai‑coustics plugin for LiveKit
|
| 12 |
+
- **[Pipecat](https://github.com/pipecat-ai/pipecat/blob/main/src/pipecat/audio/filters/aic_filter.py)** — ai‑coustics audio filter for Pipecat
|
| 13 |
+
|
| 14 |
+
[Learn more in our documentation](https://docs.ai-coustics.com/guides/models#quail-voice-focus-2-0-l-16-khz)
|
offline_pipeline.py
CHANGED
|
@@ -14,7 +14,8 @@ def retrieve_audio_information(
|
|
| 14 |
sample_id: str,
|
| 15 |
stt_model: str,
|
| 16 |
) -> tuple[str, str, str, str, str, str]:
|
| 17 |
-
|
|
|
|
| 18 |
noisy_spec_path = f"/tmp/{sample_id}_noisy_spectrogram.png"
|
| 19 |
enhanced_spec_path = f"/tmp/{sample_id}_enhanced_spectrogram.png"
|
| 20 |
spec_image(original_array).save(noisy_spec_path)
|
|
@@ -33,9 +34,12 @@ def retrieve_audio_information(
|
|
| 33 |
|
| 34 |
|
| 35 |
def denoise_audio(
|
| 36 |
-
sample_16k: np.ndarray,
|
| 37 |
enhancement_level: float = 50.0,
|
| 38 |
) -> tuple[np.ndarray | None , tuple[int, np.ndarray]| None]:
|
|
|
|
|
|
|
|
|
|
| 39 |
try:
|
| 40 |
sdk = SDKWrapper()
|
| 41 |
sdk.init_processor(sample_rate=DEFAULT_SR, enhancement_level=float(enhancement_level) / 100.0)
|
|
@@ -53,6 +57,9 @@ def load_local_file(
|
|
| 53 |
if not sample_path or not os.path.exists(sample_path):
|
| 54 |
gr.Warning("Please upload a valid audio file.")
|
| 55 |
raise ValueError("Missing audio sample. Please upload an audio sample or use the microphone input.")
|
|
|
|
|
|
|
|
|
|
| 56 |
new_sample_stem = os.path.splitext(os.path.basename(sample_path))[0]
|
| 57 |
y_16k, _ = librosa.load(sample_path, sr=DEFAULT_SR, dtype="float32", mono=True)
|
| 58 |
return y_16k, new_sample_stem
|
|
|
|
| 14 |
sample_id: str,
|
| 15 |
stt_model: str,
|
| 16 |
) -> tuple[str, str, str, str, str, str]:
|
| 17 |
+
if original_array is None or enhanced_array is None:
|
| 18 |
+
raise ValueError("Audio arrays are not available.")
|
| 19 |
noisy_spec_path = f"/tmp/{sample_id}_noisy_spectrogram.png"
|
| 20 |
enhanced_spec_path = f"/tmp/{sample_id}_enhanced_spectrogram.png"
|
| 21 |
spec_image(original_array).save(noisy_spec_path)
|
|
|
|
| 34 |
|
| 35 |
|
| 36 |
def denoise_audio(
|
| 37 |
+
sample_16k: np.ndarray,
|
| 38 |
enhancement_level: float = 50.0,
|
| 39 |
) -> tuple[np.ndarray | None , tuple[int, np.ndarray]| None]:
|
| 40 |
+
if sample_16k is None:
|
| 41 |
+
gr.Warning("No audio loaded. Please upload a file first.")
|
| 42 |
+
raise ValueError("No audio to enhance.")
|
| 43 |
try:
|
| 44 |
sdk = SDKWrapper()
|
| 45 |
sdk.init_processor(sample_rate=DEFAULT_SR, enhancement_level=float(enhancement_level) / 100.0)
|
|
|
|
| 57 |
if not sample_path or not os.path.exists(sample_path):
|
| 58 |
gr.Warning("Please upload a valid audio file.")
|
| 59 |
raise ValueError("Missing audio sample. Please upload an audio sample or use the microphone input.")
|
| 60 |
+
if os.path.getsize(sample_path) > 5 * 1024 * 1024:
|
| 61 |
+
gr.Warning("File size exceeds 5 MB limit. Please upload a smaller file.")
|
| 62 |
+
raise ValueError("Uploaded file exceeds the 5 MB size limit.")
|
| 63 |
new_sample_stem = os.path.splitext(os.path.basename(sample_path))[0]
|
| 64 |
y_16k, _ = librosa.load(sample_path, sr=DEFAULT_SR, dtype="float32", mono=True)
|
| 65 |
return y_16k, new_sample_stem
|