DRIPPY4 / app /core /audio_language_detector.py
hoangtaiii's picture
Upload 67 files
21aadae verified
Raw History Blame Contribute Delete
8.78 kB
import sys
import os
import argparse
import json
from pathlib import Path
# Automatically add nvidia DLL directories to Windows search path to resolve DLL load errors for Faster-Whisper/ctranslate2
if sys.platform == 'win32':
base_dir = Path(__file__).resolve().parent
preferred_nvidia_bins = [
"cuda_runtime",
"cublas",
"cudnn",
"cufft",
"curand",
"cusolver",
"cusparse",
"nvjitlink",
]
for parent in [base_dir] + list(base_dir.parents):
nvidia_path = parent / "env" / "Lib" / "site-packages" / "nvidia"
if nvidia_path.exists():
bin_dirs = []
for name in preferred_nvidia_bins:
p = nvidia_path / name / "bin"
if p.exists():
bin_dirs.append(p)
for bin_dir in sorted(nvidia_path.glob("*/bin")):
if bin_dir not in bin_dirs:
bin_dirs.append(bin_dir)
for bin_dir in bin_dirs:
try:
os.add_dll_directory(str(bin_dir.resolve()))
os.environ['PATH'] = str(bin_dir.resolve()) + os.pathsep + os.environ['PATH']
except Exception:
pass
break
# Import torch first to resolve nvidia dependencies and add DLL directories
try:
import torch
except ImportError:
pass
# Enforce UTF-8 for Windows console
if sys.platform == 'win32':
try:
if hasattr(sys.stdout, 'reconfigure'):
sys.stdout.reconfigure(encoding='utf-8')
if hasattr(sys.stderr, 'reconfigure'):
sys.stderr.reconfigure(encoding='utf-8')
except Exception:
pass
def is_cuda_fully_functional():
try:
import torch
if not torch.cuda.is_available():
return False
return True
except Exception:
return False
def main():
parser = argparse.ArgumentParser(description="Standalone Faster-Whisper Audio Language Detector CLI")
parser.add_argument("--audio", required=True, help="Path to input audio WAV file")
parser.add_argument("--output", required=True, help="Path to output language segments JSON file")
parser.add_argument("--model", default="base", help="Faster-Whisper model size (e.g. base, small, medium)")
parser.add_argument("--device", default="auto", choices=["cuda", "cpu", "auto"], help="Computation device")
parser.add_argument("--allow-cpu-fallback", default="true", help="Allow CPU fallback (true/false)")
args = parser.parse_args()
audio_path = Path(args.audio)
output_path = Path(args.output)
allow_cpu_fallback = args.allow_cpu_fallback.lower() in ("true", "1", "yes", "t")
if not audio_path.exists():
print(f"Error: Input audio file not found at {audio_path}", file=sys.stderr)
sys.exit(1)
print("PROGRESS: 10%", flush=True)
print("Loading faster-whisper library...")
try:
from faster_whisper import WhisperModel
import faster_whisper
import ctranslate2
print(f"[LANG DETECT] faster-whisper: {getattr(faster_whisper, '__version__', 'unknown')}")
print(f"[LANG DETECT] ctranslate2: {getattr(ctranslate2, '__version__', 'unknown')}")
print(f"[LANG DETECT] ctranslate2 cuda devices: {ctranslate2.get_cuda_device_count()}")
except ImportError as e:
print(f"Error: faster-whisper not installed. {e}", file=sys.stderr)
sys.exit(1)
device = args.device
if device == "auto":
device = "cuda" if is_cuda_fully_functional() else "cpu"
compute_type = "float16" if device == "cuda" else "int8"
# Log device selection exactly as requested
print(f"[LANG DETECT] selected device: {device}")
print(f"[LANG DETECT] compute_type: {compute_type}")
print(f"Initializing WhisperModel '{args.model}' on {device} ({compute_type})...")
try:
model = WhisperModel(args.model, device=device, compute_type=compute_type)
except Exception as e:
if device == "cuda" and not allow_cpu_fallback:
print(f"Language detection GPU requested but CUDA backend unavailable. Error: {e}", file=sys.stderr)
sys.exit(3)
print(f"Warning: Failed to load model on {device} with compute type {compute_type}: {e}", file=sys.stderr)
print("Falling back to CPU with int8...", file=sys.stderr)
try:
model = WhisperModel(args.model, device="cpu", compute_type="int8")
except Exception as ex:
print(f"Error: Failed to fallback to CPU: {ex}", file=sys.stderr)
sys.exit(1)
print("PROGRESS: 40%", flush=True)
print("Transcribing audio to detect language segments...")
try:
# Detect language segments
# vad_filter=True makes segmentation much cleaner and filters out silence/music
segments, info = model.transcribe(str(audio_path), vad_filter=True, beam_size=5)
print(f"Detected dominant language: {info.language} (probability: {info.language_probability:.2f})")
print("PROGRESS: 60%", flush=True)
result_segments = []
last_end = 0.0
for segment in segments:
if segment.start - last_end >= 1.0:
result_segments.append({
"start": round(last_end, 3),
"end": round(segment.start, 3),
"language": "music",
"text": "",
"confidence": "gap_no_speech"
})
# We want to know the language of each segment.
# In faster-whisper, segment contains the text and start/end timestamps.
# To get segment-level language, we check if the transcribing info returned is accurate.
# Wait, transcribing with faster-whisper runs on a single language detected initially.
# But wait, what if the audio has mixed languages (Chinese + English)?
# To detect mixed languages at segment level, can we run transcribe with word-level/segment-level language identification,
# or can we check if the transcribed text matches specific scripts (e.g. Chinese characters vs English words)?
# Yes! We can look at the words/characters in the transcribed segment text!
# If the segment text is >= 80% Chinese characters, we label it "zh".
# If it contains English alphabet words and no Chinese characters, we label it "en".
# This is an extremely clever, lightweight, and robust way to detect timeline languages in mixed audio!
text = segment.text.strip()
# Simple heuristic script detector:
has_chinese = bool(re.search(r'[\u4e00-\u9fff]', text))
has_english = bool(re.search(r'[a-zA-Z]', text))
# Heuristic assignment
if has_chinese and not has_english:
segment_lang = "zh"
elif has_english and not has_chinese:
segment_lang = "en"
elif has_chinese and has_english:
# Count characters to see which is dominant
chinese_chars = len(re.findall(r'[\u4e00-\u9fff]', text))
english_words = len(re.findall(r'[a-zA-Z]+', text))
if chinese_chars >= english_words:
segment_lang = "zh"
else:
segment_lang = "mixed"
else:
segment_lang = info.language if info.language in ("zh", "en") else "unknown"
result_segments.append({
"start": round(segment.start, 3),
"end": round(segment.end, 3),
"language": segment_lang,
"text": text,
"confidence": "script_heuristic"
})
last_end = max(last_end, float(segment.end))
print("PROGRESS: 90%", flush=True)
# Save to JSON
output_path.parent.mkdir(parents=True, exist_ok=True)
with open(output_path, "w", encoding="utf-8") as f:
json.dump(result_segments, f, ensure_ascii=False, indent=2)
print(f"Language detection completed. Segments saved to {output_path}")
except Exception as e:
print(f"Error during language transcription: {e}", file=sys.stderr)
sys.exit(1)
# Free VRAM/GPU Cache
try:
del model
import gc
gc.collect()
import torch
if torch.cuda.is_available():
torch.cuda.empty_cache()
except Exception:
pass
sys.exit(0)
import re # import regex here to use in heuristic
if __name__ == "__main__":
main()