Spaces:
Running on Zero
Running on Zero
File size: 8,781 Bytes
21aadae | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 | import sys
import os
import argparse
import json
from pathlib import Path
# Automatically add nvidia DLL directories to Windows search path to resolve DLL load errors for Faster-Whisper/ctranslate2
if sys.platform == 'win32':
base_dir = Path(__file__).resolve().parent
preferred_nvidia_bins = [
"cuda_runtime",
"cublas",
"cudnn",
"cufft",
"curand",
"cusolver",
"cusparse",
"nvjitlink",
]
for parent in [base_dir] + list(base_dir.parents):
nvidia_path = parent / "env" / "Lib" / "site-packages" / "nvidia"
if nvidia_path.exists():
bin_dirs = []
for name in preferred_nvidia_bins:
p = nvidia_path / name / "bin"
if p.exists():
bin_dirs.append(p)
for bin_dir in sorted(nvidia_path.glob("*/bin")):
if bin_dir not in bin_dirs:
bin_dirs.append(bin_dir)
for bin_dir in bin_dirs:
try:
os.add_dll_directory(str(bin_dir.resolve()))
os.environ['PATH'] = str(bin_dir.resolve()) + os.pathsep + os.environ['PATH']
except Exception:
pass
break
# Import torch first to resolve nvidia dependencies and add DLL directories
try:
import torch
except ImportError:
pass
# Enforce UTF-8 for Windows console
if sys.platform == 'win32':
try:
if hasattr(sys.stdout, 'reconfigure'):
sys.stdout.reconfigure(encoding='utf-8')
if hasattr(sys.stderr, 'reconfigure'):
sys.stderr.reconfigure(encoding='utf-8')
except Exception:
pass
def is_cuda_fully_functional():
try:
import torch
if not torch.cuda.is_available():
return False
return True
except Exception:
return False
def main():
parser = argparse.ArgumentParser(description="Standalone Faster-Whisper Audio Language Detector CLI")
parser.add_argument("--audio", required=True, help="Path to input audio WAV file")
parser.add_argument("--output", required=True, help="Path to output language segments JSON file")
parser.add_argument("--model", default="base", help="Faster-Whisper model size (e.g. base, small, medium)")
parser.add_argument("--device", default="auto", choices=["cuda", "cpu", "auto"], help="Computation device")
parser.add_argument("--allow-cpu-fallback", default="true", help="Allow CPU fallback (true/false)")
args = parser.parse_args()
audio_path = Path(args.audio)
output_path = Path(args.output)
allow_cpu_fallback = args.allow_cpu_fallback.lower() in ("true", "1", "yes", "t")
if not audio_path.exists():
print(f"Error: Input audio file not found at {audio_path}", file=sys.stderr)
sys.exit(1)
print("PROGRESS: 10%", flush=True)
print("Loading faster-whisper library...")
try:
from faster_whisper import WhisperModel
import faster_whisper
import ctranslate2
print(f"[LANG DETECT] faster-whisper: {getattr(faster_whisper, '__version__', 'unknown')}")
print(f"[LANG DETECT] ctranslate2: {getattr(ctranslate2, '__version__', 'unknown')}")
print(f"[LANG DETECT] ctranslate2 cuda devices: {ctranslate2.get_cuda_device_count()}")
except ImportError as e:
print(f"Error: faster-whisper not installed. {e}", file=sys.stderr)
sys.exit(1)
device = args.device
if device == "auto":
device = "cuda" if is_cuda_fully_functional() else "cpu"
compute_type = "float16" if device == "cuda" else "int8"
# Log device selection exactly as requested
print(f"[LANG DETECT] selected device: {device}")
print(f"[LANG DETECT] compute_type: {compute_type}")
print(f"Initializing WhisperModel '{args.model}' on {device} ({compute_type})...")
try:
model = WhisperModel(args.model, device=device, compute_type=compute_type)
except Exception as e:
if device == "cuda" and not allow_cpu_fallback:
print(f"Language detection GPU requested but CUDA backend unavailable. Error: {e}", file=sys.stderr)
sys.exit(3)
print(f"Warning: Failed to load model on {device} with compute type {compute_type}: {e}", file=sys.stderr)
print("Falling back to CPU with int8...", file=sys.stderr)
try:
model = WhisperModel(args.model, device="cpu", compute_type="int8")
except Exception as ex:
print(f"Error: Failed to fallback to CPU: {ex}", file=sys.stderr)
sys.exit(1)
print("PROGRESS: 40%", flush=True)
print("Transcribing audio to detect language segments...")
try:
# Detect language segments
# vad_filter=True makes segmentation much cleaner and filters out silence/music
segments, info = model.transcribe(str(audio_path), vad_filter=True, beam_size=5)
print(f"Detected dominant language: {info.language} (probability: {info.language_probability:.2f})")
print("PROGRESS: 60%", flush=True)
result_segments = []
last_end = 0.0
for segment in segments:
if segment.start - last_end >= 1.0:
result_segments.append({
"start": round(last_end, 3),
"end": round(segment.start, 3),
"language": "music",
"text": "",
"confidence": "gap_no_speech"
})
# We want to know the language of each segment.
# In faster-whisper, segment contains the text and start/end timestamps.
# To get segment-level language, we check if the transcribing info returned is accurate.
# Wait, transcribing with faster-whisper runs on a single language detected initially.
# But wait, what if the audio has mixed languages (Chinese + English)?
# To detect mixed languages at segment level, can we run transcribe with word-level/segment-level language identification,
# or can we check if the transcribed text matches specific scripts (e.g. Chinese characters vs English words)?
# Yes! We can look at the words/characters in the transcribed segment text!
# If the segment text is >= 80% Chinese characters, we label it "zh".
# If it contains English alphabet words and no Chinese characters, we label it "en".
# This is an extremely clever, lightweight, and robust way to detect timeline languages in mixed audio!
text = segment.text.strip()
# Simple heuristic script detector:
has_chinese = bool(re.search(r'[\u4e00-\u9fff]', text))
has_english = bool(re.search(r'[a-zA-Z]', text))
# Heuristic assignment
if has_chinese and not has_english:
segment_lang = "zh"
elif has_english and not has_chinese:
segment_lang = "en"
elif has_chinese and has_english:
# Count characters to see which is dominant
chinese_chars = len(re.findall(r'[\u4e00-\u9fff]', text))
english_words = len(re.findall(r'[a-zA-Z]+', text))
if chinese_chars >= english_words:
segment_lang = "zh"
else:
segment_lang = "mixed"
else:
segment_lang = info.language if info.language in ("zh", "en") else "unknown"
result_segments.append({
"start": round(segment.start, 3),
"end": round(segment.end, 3),
"language": segment_lang,
"text": text,
"confidence": "script_heuristic"
})
last_end = max(last_end, float(segment.end))
print("PROGRESS: 90%", flush=True)
# Save to JSON
output_path.parent.mkdir(parents=True, exist_ok=True)
with open(output_path, "w", encoding="utf-8") as f:
json.dump(result_segments, f, ensure_ascii=False, indent=2)
print(f"Language detection completed. Segments saved to {output_path}")
except Exception as e:
print(f"Error during language transcription: {e}", file=sys.stderr)
sys.exit(1)
# Free VRAM/GPU Cache
try:
del model
import gc
gc.collect()
import torch
if torch.cuda.is_available():
torch.cuda.empty_cache()
except Exception:
pass
sys.exit(0)
import re # import regex here to use in heuristic
if __name__ == "__main__":
main()
|