Spaces:
Running on Zero
Running on Zero
Download app/core/audio_language_detector.py from hoangtaiii/DRIPPY4: direct link, hf CLI and curl.
- Browser
- Download file 8.78 kB
-
https://huggingface.co/spaces/hoangtaiii/DRIPPY4/resolve/main/app/core/audio_language_detector.py
- Command line
-
hf download hf://spaces/hoangtaiii/DRIPPY4/app/core/audio_language_detector.py
-
curl -L -o audio_language_detector.py https://huggingface.co/spaces/hoangtaiii/DRIPPY4/resolve/main/app/core/audio_language_detector.py
8.78 kB
| import sys | |
| import os | |
| import argparse | |
| import json | |
| from pathlib import Path | |
| # Automatically add nvidia DLL directories to Windows search path to resolve DLL load errors for Faster-Whisper/ctranslate2 | |
| if sys.platform == 'win32': | |
| base_dir = Path(__file__).resolve().parent | |
| preferred_nvidia_bins = [ | |
| "cuda_runtime", | |
| "cublas", | |
| "cudnn", | |
| "cufft", | |
| "curand", | |
| "cusolver", | |
| "cusparse", | |
| "nvjitlink", | |
| ] | |
| for parent in [base_dir] + list(base_dir.parents): | |
| nvidia_path = parent / "env" / "Lib" / "site-packages" / "nvidia" | |
| if nvidia_path.exists(): | |
| bin_dirs = [] | |
| for name in preferred_nvidia_bins: | |
| p = nvidia_path / name / "bin" | |
| if p.exists(): | |
| bin_dirs.append(p) | |
| for bin_dir in sorted(nvidia_path.glob("*/bin")): | |
| if bin_dir not in bin_dirs: | |
| bin_dirs.append(bin_dir) | |
| for bin_dir in bin_dirs: | |
| try: | |
| os.add_dll_directory(str(bin_dir.resolve())) | |
| os.environ['PATH'] = str(bin_dir.resolve()) + os.pathsep + os.environ['PATH'] | |
| except Exception: | |
| pass | |
| break | |
| # Import torch first to resolve nvidia dependencies and add DLL directories | |
| try: | |
| import torch | |
| except ImportError: | |
| pass | |
| # Enforce UTF-8 for Windows console | |
| if sys.platform == 'win32': | |
| try: | |
| if hasattr(sys.stdout, 'reconfigure'): | |
| sys.stdout.reconfigure(encoding='utf-8') | |
| if hasattr(sys.stderr, 'reconfigure'): | |
| sys.stderr.reconfigure(encoding='utf-8') | |
| except Exception: | |
| pass | |
| def is_cuda_fully_functional(): | |
| try: | |
| import torch | |
| if not torch.cuda.is_available(): | |
| return False | |
| return True | |
| except Exception: | |
| return False | |
| def main(): | |
| parser = argparse.ArgumentParser(description="Standalone Faster-Whisper Audio Language Detector CLI") | |
| parser.add_argument("--audio", required=True, help="Path to input audio WAV file") | |
| parser.add_argument("--output", required=True, help="Path to output language segments JSON file") | |
| parser.add_argument("--model", default="base", help="Faster-Whisper model size (e.g. base, small, medium)") | |
| parser.add_argument("--device", default="auto", choices=["cuda", "cpu", "auto"], help="Computation device") | |
| parser.add_argument("--allow-cpu-fallback", default="true", help="Allow CPU fallback (true/false)") | |
| args = parser.parse_args() | |
| audio_path = Path(args.audio) | |
| output_path = Path(args.output) | |
| allow_cpu_fallback = args.allow_cpu_fallback.lower() in ("true", "1", "yes", "t") | |
| if not audio_path.exists(): | |
| print(f"Error: Input audio file not found at {audio_path}", file=sys.stderr) | |
| sys.exit(1) | |
| print("PROGRESS: 10%", flush=True) | |
| print("Loading faster-whisper library...") | |
| try: | |
| from faster_whisper import WhisperModel | |
| import faster_whisper | |
| import ctranslate2 | |
| print(f"[LANG DETECT] faster-whisper: {getattr(faster_whisper, '__version__', 'unknown')}") | |
| print(f"[LANG DETECT] ctranslate2: {getattr(ctranslate2, '__version__', 'unknown')}") | |
| print(f"[LANG DETECT] ctranslate2 cuda devices: {ctranslate2.get_cuda_device_count()}") | |
| except ImportError as e: | |
| print(f"Error: faster-whisper not installed. {e}", file=sys.stderr) | |
| sys.exit(1) | |
| device = args.device | |
| if device == "auto": | |
| device = "cuda" if is_cuda_fully_functional() else "cpu" | |
| compute_type = "float16" if device == "cuda" else "int8" | |
| # Log device selection exactly as requested | |
| print(f"[LANG DETECT] selected device: {device}") | |
| print(f"[LANG DETECT] compute_type: {compute_type}") | |
| print(f"Initializing WhisperModel '{args.model}' on {device} ({compute_type})...") | |
| try: | |
| model = WhisperModel(args.model, device=device, compute_type=compute_type) | |
| except Exception as e: | |
| if device == "cuda" and not allow_cpu_fallback: | |
| print(f"Language detection GPU requested but CUDA backend unavailable. Error: {e}", file=sys.stderr) | |
| sys.exit(3) | |
| print(f"Warning: Failed to load model on {device} with compute type {compute_type}: {e}", file=sys.stderr) | |
| print("Falling back to CPU with int8...", file=sys.stderr) | |
| try: | |
| model = WhisperModel(args.model, device="cpu", compute_type="int8") | |
| except Exception as ex: | |
| print(f"Error: Failed to fallback to CPU: {ex}", file=sys.stderr) | |
| sys.exit(1) | |
| print("PROGRESS: 40%", flush=True) | |
| print("Transcribing audio to detect language segments...") | |
| try: | |
| # Detect language segments | |
| # vad_filter=True makes segmentation much cleaner and filters out silence/music | |
| segments, info = model.transcribe(str(audio_path), vad_filter=True, beam_size=5) | |
| print(f"Detected dominant language: {info.language} (probability: {info.language_probability:.2f})") | |
| print("PROGRESS: 60%", flush=True) | |
| result_segments = [] | |
| last_end = 0.0 | |
| for segment in segments: | |
| if segment.start - last_end >= 1.0: | |
| result_segments.append({ | |
| "start": round(last_end, 3), | |
| "end": round(segment.start, 3), | |
| "language": "music", | |
| "text": "", | |
| "confidence": "gap_no_speech" | |
| }) | |
| # We want to know the language of each segment. | |
| # In faster-whisper, segment contains the text and start/end timestamps. | |
| # To get segment-level language, we check if the transcribing info returned is accurate. | |
| # Wait, transcribing with faster-whisper runs on a single language detected initially. | |
| # But wait, what if the audio has mixed languages (Chinese + English)? | |
| # To detect mixed languages at segment level, can we run transcribe with word-level/segment-level language identification, | |
| # or can we check if the transcribed text matches specific scripts (e.g. Chinese characters vs English words)? | |
| # Yes! We can look at the words/characters in the transcribed segment text! | |
| # If the segment text is >= 80% Chinese characters, we label it "zh". | |
| # If it contains English alphabet words and no Chinese characters, we label it "en". | |
| # This is an extremely clever, lightweight, and robust way to detect timeline languages in mixed audio! | |
| text = segment.text.strip() | |
| # Simple heuristic script detector: | |
| has_chinese = bool(re.search(r'[\u4e00-\u9fff]', text)) | |
| has_english = bool(re.search(r'[a-zA-Z]', text)) | |
| # Heuristic assignment | |
| if has_chinese and not has_english: | |
| segment_lang = "zh" | |
| elif has_english and not has_chinese: | |
| segment_lang = "en" | |
| elif has_chinese and has_english: | |
| # Count characters to see which is dominant | |
| chinese_chars = len(re.findall(r'[\u4e00-\u9fff]', text)) | |
| english_words = len(re.findall(r'[a-zA-Z]+', text)) | |
| if chinese_chars >= english_words: | |
| segment_lang = "zh" | |
| else: | |
| segment_lang = "mixed" | |
| else: | |
| segment_lang = info.language if info.language in ("zh", "en") else "unknown" | |
| result_segments.append({ | |
| "start": round(segment.start, 3), | |
| "end": round(segment.end, 3), | |
| "language": segment_lang, | |
| "text": text, | |
| "confidence": "script_heuristic" | |
| }) | |
| last_end = max(last_end, float(segment.end)) | |
| print("PROGRESS: 90%", flush=True) | |
| # Save to JSON | |
| output_path.parent.mkdir(parents=True, exist_ok=True) | |
| with open(output_path, "w", encoding="utf-8") as f: | |
| json.dump(result_segments, f, ensure_ascii=False, indent=2) | |
| print(f"Language detection completed. Segments saved to {output_path}") | |
| except Exception as e: | |
| print(f"Error during language transcription: {e}", file=sys.stderr) | |
| sys.exit(1) | |
| # Free VRAM/GPU Cache | |
| try: | |
| del model | |
| import gc | |
| gc.collect() | |
| import torch | |
| if torch.cuda.is_available(): | |
| torch.cuda.empty_cache() | |
| except Exception: | |
| pass | |
| sys.exit(0) | |
| import re # import regex here to use in heuristic | |
| if __name__ == "__main__": | |
| main() | |