lumicollect / core /audio_verifier.py
nemabruh404's picture
Rename ui/audio_verifier.py to core/audio_verifier.py
2aff6e0 verified
Raw
History Blame
2.16 kB
import numpy as np
from config import SCORE_THRESHOLD, CHUNK_SIZE
from core.model_loader import wait_until_ready, get_model, get_error
from utils.audio_utils import load_audio_as_int16, iter_chunks, make_silence
# Số chunk silence chạy sau khi verify xong
# để flush state bên trong ONNX model,
# tránh ảnh hưởng sang lần verify tiếp theo
_FLUSH_CHUNKS = 7
def _run_model(audio_int16: np.ndarray) -> float:
"""
Chạy model trên toàn bộ audio, trả về score cao nhất tìm được.
Gọi hàm này chỉ khi đã chắc chắn model sẵn sàng.
"""
model = get_model()
model.reset()
max_score = 0.0
for chunk in iter_chunks(audio_int16):
prediction = model.predict(chunk)
raw = list(prediction.values())[0]
score = float(raw[0]) if isinstance(raw, (np.ndarray, list)) else float(raw)
if score > max_score:
max_score = score
# Flush model state
silence = make_silence()
for _ in range(_FLUSH_CHUNKS):
model.predict(silence)
return max_score
def verify_audio(audio_path: str) -> tuple[bool, float]:
"""
Kiểm tra chất lượng mẫu âm thanh bằng wakeword model.
Returns:
(passed, score)
passed — True nếu score >= SCORE_THRESHOLD
score — float 0.0 ~ 1.0
Raises:
RuntimeError — model chưa sẵn sàng hoặc bị lỗi lúc load
ValueError — file audio không đọc được
"""
# --- Kiểm tra model ---
if not wait_until_ready():
error = get_error()
if error:
raise RuntimeError(f"Model lỗi khi khởi động: {error}")
raise RuntimeError("Model chưa sẵn sàng, vui lòng thử lại sau.")
# --- Kiểm tra file ---
if not audio_path:
raise ValueError("Không có file audio.")
# --- Load audio ---
try:
audio_int16 = load_audio_as_int16(audio_path)
except Exception as e:
raise ValueError(f"Không đọc được file audio: {e}")
# --- Chạy model ---
score = _run_model(audio_int16)
return score >= SCORE_THRESHOLD, score