import sys import os import argparse import json from pathlib import Path # Automatically add nvidia DLL directories to Windows search path to resolve DLL load errors for Faster-Whisper/ctranslate2 if sys.platform == 'win32': base_dir = Path(__file__).resolve().parent preferred_nvidia_bins = [ "cuda_runtime", "cublas", "cudnn", "cufft", "curand", "cusolver", "cusparse", "nvjitlink", ] for parent in [base_dir] + list(base_dir.parents): nvidia_path = parent / "env" / "Lib" / "site-packages" / "nvidia" if nvidia_path.exists(): bin_dirs = [] for name in preferred_nvidia_bins: p = nvidia_path / name / "bin" if p.exists(): bin_dirs.append(p) for bin_dir in sorted(nvidia_path.glob("*/bin")): if bin_dir not in bin_dirs: bin_dirs.append(bin_dir) for bin_dir in bin_dirs: try: os.add_dll_directory(str(bin_dir.resolve())) os.environ['PATH'] = str(bin_dir.resolve()) + os.pathsep + os.environ['PATH'] except Exception: pass break # Import torch first to resolve nvidia dependencies and add DLL directories try: import torch except ImportError: pass # Enforce UTF-8 for Windows console if sys.platform == 'win32': try: if hasattr(sys.stdout, 'reconfigure'): sys.stdout.reconfigure(encoding='utf-8') if hasattr(sys.stderr, 'reconfigure'): sys.stderr.reconfigure(encoding='utf-8') except Exception: pass def is_cuda_fully_functional(): try: import torch if not torch.cuda.is_available(): return False return True except Exception: return False def main(): parser = argparse.ArgumentParser(description="Standalone Faster-Whisper Audio Language Detector CLI") parser.add_argument("--audio", required=True, help="Path to input audio WAV file") parser.add_argument("--output", required=True, help="Path to output language segments JSON file") parser.add_argument("--model", default="base", help="Faster-Whisper model size (e.g. base, small, medium)") parser.add_argument("--device", default="auto", choices=["cuda", "cpu", "auto"], help="Computation device") parser.add_argument("--allow-cpu-fallback", default="true", help="Allow CPU fallback (true/false)") args = parser.parse_args() audio_path = Path(args.audio) output_path = Path(args.output) allow_cpu_fallback = args.allow_cpu_fallback.lower() in ("true", "1", "yes", "t") if not audio_path.exists(): print(f"Error: Input audio file not found at {audio_path}", file=sys.stderr) sys.exit(1) print("PROGRESS: 10%", flush=True) print("Loading faster-whisper library...") try: from faster_whisper import WhisperModel import faster_whisper import ctranslate2 print(f"[LANG DETECT] faster-whisper: {getattr(faster_whisper, '__version__', 'unknown')}") print(f"[LANG DETECT] ctranslate2: {getattr(ctranslate2, '__version__', 'unknown')}") print(f"[LANG DETECT] ctranslate2 cuda devices: {ctranslate2.get_cuda_device_count()}") except ImportError as e: print(f"Error: faster-whisper not installed. {e}", file=sys.stderr) sys.exit(1) device = args.device if device == "auto": device = "cuda" if is_cuda_fully_functional() else "cpu" compute_type = "float16" if device == "cuda" else "int8" # Log device selection exactly as requested print(f"[LANG DETECT] selected device: {device}") print(f"[LANG DETECT] compute_type: {compute_type}") print(f"Initializing WhisperModel '{args.model}' on {device} ({compute_type})...") try: model = WhisperModel(args.model, device=device, compute_type=compute_type) except Exception as e: if device == "cuda" and not allow_cpu_fallback: print(f"Language detection GPU requested but CUDA backend unavailable. Error: {e}", file=sys.stderr) sys.exit(3) print(f"Warning: Failed to load model on {device} with compute type {compute_type}: {e}", file=sys.stderr) print("Falling back to CPU with int8...", file=sys.stderr) try: model = WhisperModel(args.model, device="cpu", compute_type="int8") except Exception as ex: print(f"Error: Failed to fallback to CPU: {ex}", file=sys.stderr) sys.exit(1) print("PROGRESS: 40%", flush=True) print("Transcribing audio to detect language segments...") try: # Detect language segments # vad_filter=True makes segmentation much cleaner and filters out silence/music segments, info = model.transcribe(str(audio_path), vad_filter=True, beam_size=5) print(f"Detected dominant language: {info.language} (probability: {info.language_probability:.2f})") print("PROGRESS: 60%", flush=True) result_segments = [] last_end = 0.0 for segment in segments: if segment.start - last_end >= 1.0: result_segments.append({ "start": round(last_end, 3), "end": round(segment.start, 3), "language": "music", "text": "", "confidence": "gap_no_speech" }) # We want to know the language of each segment. # In faster-whisper, segment contains the text and start/end timestamps. # To get segment-level language, we check if the transcribing info returned is accurate. # Wait, transcribing with faster-whisper runs on a single language detected initially. # But wait, what if the audio has mixed languages (Chinese + English)? # To detect mixed languages at segment level, can we run transcribe with word-level/segment-level language identification, # or can we check if the transcribed text matches specific scripts (e.g. Chinese characters vs English words)? # Yes! We can look at the words/characters in the transcribed segment text! # If the segment text is >= 80% Chinese characters, we label it "zh". # If it contains English alphabet words and no Chinese characters, we label it "en". # This is an extremely clever, lightweight, and robust way to detect timeline languages in mixed audio! text = segment.text.strip() # Simple heuristic script detector: has_chinese = bool(re.search(r'[\u4e00-\u9fff]', text)) has_english = bool(re.search(r'[a-zA-Z]', text)) # Heuristic assignment if has_chinese and not has_english: segment_lang = "zh" elif has_english and not has_chinese: segment_lang = "en" elif has_chinese and has_english: # Count characters to see which is dominant chinese_chars = len(re.findall(r'[\u4e00-\u9fff]', text)) english_words = len(re.findall(r'[a-zA-Z]+', text)) if chinese_chars >= english_words: segment_lang = "zh" else: segment_lang = "mixed" else: segment_lang = info.language if info.language in ("zh", "en") else "unknown" result_segments.append({ "start": round(segment.start, 3), "end": round(segment.end, 3), "language": segment_lang, "text": text, "confidence": "script_heuristic" }) last_end = max(last_end, float(segment.end)) print("PROGRESS: 90%", flush=True) # Save to JSON output_path.parent.mkdir(parents=True, exist_ok=True) with open(output_path, "w", encoding="utf-8") as f: json.dump(result_segments, f, ensure_ascii=False, indent=2) print(f"Language detection completed. Segments saved to {output_path}") except Exception as e: print(f"Error during language transcription: {e}", file=sys.stderr) sys.exit(1) # Free VRAM/GPU Cache try: del model import gc gc.collect() import torch if torch.cuda.is_available(): torch.cuda.empty_cache() except Exception: pass sys.exit(0) import re # import regex here to use in heuristic if __name__ == "__main__": main()