""" app/core/cloud_pipeline.py ────────────────────────── Pure Cloud-Native Headless Pipeline Coordinator. Features: 1. High-Definition Video Preview Frame Generator using FFmpeg (100% Reliable). 2. Multi-Model ASR Matrix (Groq Whisper Large V3 / OpenRouter / Gemini). 3. AI Vision Models OCR with automatic ASR fallback. 4. Active 2026 AI Translation Models (Groq Qwen 3.6 / GPT-OSS 120B / Gemini Flash / Nemotron 3). 5. Glossary & Slang Mapping (Fashion, Sneaker, Streetwear, Chinese Slang). 6. 7-Step Quality Guard & Semantic Validation & Timing Compaction. 7. Subtitle Editing Hooks (extract_subtitles_only & render_from_subtitles). 8. ASS Subtitle Engine with TikTok / Shorts Typography & Word Jump Styles. 9. 100% Solid/Delogo Masking Box to eliminate old Chinese subtitles. 10. Studio SFX & Audio Ducking Mixer. 11. SQLite JobManager Integration for state persistence. """ import os import re import sys import json import time import base64 import shutil import cv2 import requests import subprocess from pathlib import Path from typing import Optional, Callable, Dict, Any, Tuple, List from app.core.cloud_asr import CloudASREngine from app.core.cloud_ocr import CloudOCREngine from app.core.cloud_tts import CloudTTSEngine from app.core.vietnamese_text_normalizer import VietnameseTextNormalizer from app.core.translation_validator import TranslationValidator from app.core.translation_quality_guard import TranslationQualityGuard from app.core.translation_post_editor import post_edit_translation from app.core.google_translate_fallback import GoogleTranslateFallback from app.core.subtitle_compactor import compact_blocks_for_tts from app.core.job_manager import JobManager from app.core.studio_sfx import generate_sfx_clip def has_chinese(text: str) -> bool: """Returns True if string contains CJK Chinese characters.""" return bool(re.search(r"[\u4e00-\u9fff]", text)) class CloudPipeline: def __init__( self, base_dir: Optional[Path] = None, log_callback: Optional[Callable[[str], None]] = None, progress_callback: Optional[Callable[[int, str], None]] = None, ffmpeg_path: str = "ffmpeg" ): self.base_dir = Path(base_dir or Path(__file__).resolve().parents[2]) self.output_dir = self.base_dir / "output" self.temp_dir = self.base_dir / "temp" self.output_dir.mkdir(parents=True, exist_ok=True) self.temp_dir.mkdir(parents=True, exist_ok=True) self.log_fn = log_callback or print self.progress_fn = progress_callback or (lambda pct, stage: None) self.ffmpeg_path = ffmpeg_path self.asr_engine = CloudASREngine(log_fn=self._log) self.ocr_engine = CloudOCREngine(log_fn=self._log) self.tts_engine = CloudTTSEngine(log_fn=self._log, ffmpeg_path=self.ffmpeg_path) self.normalizer = VietnameseTextNormalizer() self.job_manager = JobManager.instance() self.glossary = self._load_glossary() def _log(self, msg: str): self.log_fn(f"[Cloud Pipeline] {msg}") def _get_video_duration_sec(self, v_path: Path) -> float: """Probe video duration (sec). ffprobe -> cv2 fallback. 0.0 if unknown.""" try: import shutil as _sh _ffprobe = _sh.which("ffprobe") if not _ffprobe: cand = Path(self.ffmpeg_path).parent / "ffprobe.exe" _ffprobe = str(cand) if cand.exists() else "ffprobe" _res = subprocess.run( [_ffprobe, "-v", "error", "-show_entries", "format=duration", "-of", "default=noprint_wrappers=1:nokey=1", str(v_path)], capture_output=True, text=True, timeout=15) _dur = float((_res.stdout or "").strip()) if _dur > 0: return _dur except Exception: pass try: cap = cv2.VideoCapture(str(v_path)) fps = cap.get(cv2.CAP_PROP_FPS) or 0 frames = cap.get(cv2.CAP_PROP_FRAME_COUNT) or 0 cap.release() if fps > 0 and frames > 0: return float(frames / fps) except Exception: pass return 0.0 def _pad_audio_to_video_duration(self, audio_path: Path, video_dur: float) -> bool: """Pad (never cut) audio to exactly video duration. Guards against `-shortest` slicing the video when dubbing/mix is shorter than source (e.g. mute_original + truncated subtitles: 2min video -> 20s output).""" if video_dur <= 0 or not audio_path.exists(): return False try: tmp = audio_path.parent / f"{audio_path.stem}_padded.wav" cmd = [str(self.ffmpeg_path), "-y", "-i", str(audio_path), "-filter:a", f"apad=whole_dur={video_dur:.3f}", "-ac", "2", "-ar", "48000", "-t", f"{video_dur:.3f}", str(tmp)] res = subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.PIPE, timeout=120) if res.returncode == 0 and tmp.exists() and tmp.stat().st_size > 1000: tmp.replace(audio_path) return True self._log(f"⚠️ Pad audio to {video_dur:.1f}s failed, keeping original mix.") try: if tmp.exists(): tmp.unlink() except Exception: pass except Exception as e: self._log(f"⚠️ Pad audio error: {e}") return False def _load_glossary(self) -> Dict[str, str]: glossary_path = self.base_dir / "glossary.json" if glossary_path.exists(): try: data = json.loads(glossary_path.read_text(encoding="utf-8")) return data.get("glossary", {}) except Exception as e: self._log(f"⚠️ Lỗi đọc glossary.json: {e}") return {} def calculate_region( self, video_path: Path, region_preset: str = "custom", custom_y_pct: float = 75.0, custom_h_pct: float = 20.0, custom_x_pct: float = 0.0, custom_w_pct: float = 100.0 ) -> Tuple[int, int, int, int]: vw, vh = 1920, 1080 try: cap = cv2.VideoCapture(str(video_path)) w = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH)) h = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT)) cap.release() if w > 0 and h > 0: vw, vh = w, h except Exception: pass if region_preset == "bottom_25": x = 0 w = vw y = int(vh * 0.72) h = int(vh * 0.24) elif region_preset == "bottom_15": x = 0 w = vw y = int(vh * 0.82) h = int(vh * 0.15) elif region_preset == "middle_25": x = 0 w = vw y = int(vh * 0.38) h = int(vh * 0.24) elif region_preset == "none": return (0, 0, 0, 0) elif region_preset == "custom": x = int(vw * (custom_x_pct / 100.0)) w = int(vw * (custom_w_pct / 100.0)) y = int(vh * (custom_y_pct / 100.0)) h = int(vh * (custom_h_pct / 100.0)) else: return (0, 0, 0, 0) x = max(0, min(vw - 2, x)) y = max(0, min(vh - 2, y)) w = max(2, min(vw - x, w)) h = max(2, min(vh - y, h)) return (x, y, w, h) def generate_preview_frame( self, video_path: str, region_preset: str = "custom", custom_y_pct: float = 75.0, custom_h_pct: float = 20.0, custom_x_pct: float = 0.0, custom_w_pct: float = 100.0, sec: float = 2.0, return_type: str = "rgb", # ── Logo overlay params ── logo_enabled: bool = False, logo_path: str = "", logo_preset: str = "bottom_right", logo_scale: float = 15.0, logo_opacity: float = 1.0, logo_x_pct: float = 80.0, logo_y_pct: float = 80.0, logo_chromakey: bool = True, # ── CTA video overlay params (appears every 5s) ── cta_enabled: bool = False, cta_path: str = "", cta_preset: str = "bottom_center", cta_scale: float = 35.0, cta_opacity: float = 1.0, cta_x_pct: float = 50.0, cta_y_pct: float = 85.0, cta_interval: float = 5.0, cta_duration: float = 2.0, cta_chromakey: bool = True ): """Extracts a frame at timestamp sec and overlays bounding box + logo + CTA preview.""" v_file = Path(video_path) if not v_file.exists(): return None raw_frame_path = self.temp_dir / f"raw_frame_{int(time.time()*1000)}.jpg" cmd = [ str(self.ffmpeg_path), "-ss", str(sec), "-y", "-i", str(v_file), "-frames:v", "1", "-q:v", "2", str(raw_frame_path) ] try: subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True) frame = cv2.imread(str(raw_frame_path)) if raw_frame_path.exists(): raw_frame_path.unlink() except Exception: frame = None if frame is None: cap = cv2.VideoCapture(str(v_file)) fps = cap.get(cv2.CAP_PROP_FPS) or 25.0 target_frame = max(0, int(sec * fps)) cap.set(cv2.CAP_PROP_POS_FRAMES, target_frame) ret, frame = cap.read() cap.release() if frame is None: return None x, y, w, h = self.calculate_region( v_file, region_preset, custom_y_pct, custom_h_pct, custom_x_pct, custom_w_pct ) if w > 0 and h > 0: overlay = frame.copy() cv2.rectangle(overlay, (x, y), (x + w, y + h), (0, 0, 255), -1) cv2.addWeighted(overlay, 0.35, frame, 0.65, 0, frame) cv2.rectangle(frame, (x, y), (x + w, y + h), (0, 255, 255), 3) label = "VUNG CHE SUB CU & QUET OCR" font = cv2.FONT_HERSHEY_SIMPLEX font_scale = max(0.5, frame.shape[1] / 1200.0) thickness = 2 (tw, th), _ = cv2.getTextSize(label, font, font_scale, thickness) label_y = max(th + 10, y - 8) cv2.rectangle(frame, (x, label_y - th - 6), (x + tw + 10, label_y + 4), (0, 0, 0), -1) cv2.putText(frame, label, (x + 5, label_y), font, font_scale, (0, 255, 255), thickness, cv2.LINE_AA) # ── Logo overlay preview ── if logo_enabled and logo_path and Path(logo_path).exists(): try: from app.core.logo_overlay import load_logo_rgba, overlay_logo_on_frame _logo_rgba = load_logo_rgba(logo_path, chromakey=logo_chromakey) if _logo_rgba is not None: frame = overlay_logo_on_frame( frame_bgr=frame, logo_rgba=_logo_rgba, preset=logo_preset, scale_pct=float(logo_scale), opacity=float(logo_opacity), custom_x_pct=float(logo_x_pct), custom_y_pct=float(logo_y_pct), margin_pct=2.0 ) cv2.putText(frame, f"LOGO:{logo_preset} {logo_scale:.0f}%", (10, 30), cv2.FONT_HERSHEY_SIMPLEX, 0.7, (0, 255, 255), 2, cv2.LINE_AA) except Exception as _e: self._log(f"⚠️ Logo preview error: {_e}") # ── CTA video overlay preview (every 5s) ── if cta_enabled and cta_path and Path(cta_path).exists(): try: from app.core.logo_overlay import load_cta_frame_rgba, overlay_cta_on_frame, should_show_cta_at_time if should_show_cta_at_time(float(sec), float(cta_interval), float(cta_duration)): # CTA time loops within CTA video duration _cta_time = float(sec) % float(cta_interval) # If CTA video is shorter than interval, loop inside _cta_rgba = load_cta_frame_rgba(cta_path, cta_time_sec=_cta_time, chromakey=cta_chromakey) if _cta_rgba is not None: frame = overlay_cta_on_frame( frame_bgr=frame, cta_rgba=_cta_rgba, preset=cta_preset, scale_pct=float(cta_scale), opacity=float(cta_opacity), custom_x_pct=float(cta_x_pct), custom_y_pct=float(cta_y_pct), margin_pct=2.0 ) cv2.putText(frame, f"CTA:{cta_preset} {cta_scale:.0f}% every {cta_interval:.0f}s", (10, 60), cv2.FONT_HERSHEY_SIMPLEX, 0.6, (0, 255, 0), 2, cv2.LINE_AA) else: cv2.putText(frame, f"CTA: hidden at {sec:.1f}s (interval {cta_interval:.0f}s)", (10, 60), cv2.FONT_HERSHEY_SIMPLEX, 0.5, (0, 165, 255), 1, cv2.LINE_AA) except Exception as _e: self._log(f"⚠️ CTA preview error: {_e}") if return_type == "base64": _, buffer = cv2.imencode('.jpg', frame, [cv2.IMWRITE_JPEG_QUALITY, 85]) return base64.b64encode(buffer).decode('utf-8') return cv2.cvtColor(frame, cv2.COLOR_BGR2RGB) def extract_subtitles_only( self, video_path: str, mode: str = "asr", source_lang: str = "zh", region_preset: str = "custom", custom_y_pct: float = 75.0, custom_h_pct: float = 20.0, custom_x_pct: float = 0.0, custom_w_pct: float = 100.0 ) -> Optional[Dict[str, Any]]: """ Runs Stage 0 -> Stage A -> Stage C. Returns parsed subtitle blocks (original & translated) for Web Subtitle Editor. """ v_path = Path(video_path) if not v_path.exists(): self._log(f"❌ Video not found: {video_path}") return None video_stem = v_path.stem vtd = self.temp_dir / video_stem vtd.mkdir(parents=True, exist_ok=True) self.progress_fn(5, "STAGE_0_PREPARE") calc_region = self.calculate_region( v_path, region_preset, custom_y_pct, custom_h_pct, custom_x_pct, custom_w_pct ) x, y, w, h = calc_region # 1. Extract audio extracted_audio = vtd / "extracted_audio.wav" self._log("⚡ Trích xuất âm thanh từ video gốc...") cmd = [ str(self.ffmpeg_path), "-y", "-i", str(v_path), "-vn", "-acodec", "pcm_s16le", "-ar", "16000", "-ac", "1", str(extracted_audio) ] subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True) # 2. OCR / ASR self.progress_fn(25, "STAGE_A_OCR_ASR") original_srt = vtd / "original.srt" if mode == "ocr": lang_label = "Tiếng Trung 🇨🇳" if str(source_lang).lower().startswith("zh") else "Tiếng Anh 🇺🇸" self._log(f"👁️ Quét chữ phụ đề bằng AI Vision Models (ưu tiên {lang_label} - source_lang={source_lang})...") ocr_box = (x, y, w, h) if (w > 0 and h > 0) else None ok = self.ocr_engine.scan_video_subtitles_to_srt(str(v_path), str(original_srt), blur_region=ocr_box, source_lang=source_lang) if not ok or not original_srt.exists() or original_srt.stat().st_size < 10: # Report detailed reason before fallback reason = [] if not ok: reason.append("Vision API trả về rỗng / lỗi key") if not original_srt.exists(): reason.append("SRT file không tồn tại") elif original_srt.stat().st_size < 10: reason.append(f"SRT quá nhỏ ({original_srt.stat().st_size} bytes)") self._log(f"⚠️ OCR không phát hiện chữ ({', '.join(reason)}) -> Tự động chuyển sang Cloud ASR (fallback)...") self._log(f" 💡 Kiểm tra: 1) API Key Gemini/OpenRouter còn hạn? 2) Vùng cắt [{x},{y},{w},{h}] có đúng chứa sub? 3) Thử tăng độ nhạy hoặc chọn preset Đáy Màn Hình") ok = self.asr_engine.transcribe_audio_to_srt(str(extracted_audio), str(original_srt), source_lang=source_lang) else: self._log("🎙️ Nhận diện giọng nói bằng Cloud ASR (Groq Whisper Large-v3)...") ok = self.asr_engine.transcribe_audio_to_srt(str(extracted_audio), str(original_srt), source_lang=source_lang) if not ok or not original_srt.exists() or original_srt.stat().st_size < 10: raise RuntimeError("Không thể trích xuất phụ đề từ video.") # 3. Translation self.progress_fn(50, "STAGE_C_TRANSLATION") self._log("🌐 Dịch thuật bằng AI Translation Matrix...") translated_srt = vtd / "translated.srt" self._translate_srt_cloud(original_srt, translated_srt, source_lang=source_lang) orig_blocks = self._parse_srt_blocks(original_srt.read_text(encoding="utf-8", errors="ignore")) trans_blocks = self._parse_srt_blocks(translated_srt.read_text(encoding="utf-8", errors="ignore")) combined = [] for ob in orig_blocks: tb = next((t for t in trans_blocks if t["id"] == ob["id"]), None) combined.append({ "id": ob["id"], "timing": ob["timing"], "start_ms": ob["start_ms"], "end_ms": ob["end_ms"], "original_text": ob["text"], "vietnamese_text": tb["text"] if tb else ob["text"], "sfx": "" }) return { "video_stem": video_stem, "blocks": combined, "original_srt": str(original_srt), "translated_srt": str(translated_srt) } def render_from_subtitles( self, video_path: str, subtitles_data: List[Dict[str, Any]], voice: str = "vi-VN-NamMinhNeural", speed: float = 1.0, pitch: int = 0, volume: int = 100, region_preset: str = "custom", sub_mask_mode: str = "box", sub_style: str = "motion_drip", custom_y_pct: float = 75.0, custom_h_pct: float = 20.0, custom_x_pct: float = 0.0, custom_w_pct: float = 100.0, ducking_ratio: float = 0.18, enable_sfx: bool = True, mute_original_audio: bool = False, # ── Logo overlay ── logo_enabled: bool = False, logo_path: str = "", logo_preset: str = "bottom_right", logo_scale: float = 15.0, logo_opacity: float = 1.0, logo_x_pct: float = 80.0, logo_y_pct: float = 80.0, logo_chromakey: bool = True, # ── CTA video overlay (every 5s) ── cta_enabled: bool = False, cta_path: str = "", cta_preset: str = "bottom_center", cta_scale: float = 35.0, cta_opacity: float = 1.0, cta_x_pct: float = 50.0, cta_y_pct: float = 85.0, cta_interval: float = 5.0, cta_duration: float = 2.0, cta_chromakey: bool = True ) -> Optional[str]: """Completes TTS synthesis, Ducking audio mix, and video rendering from user-reviewed subtitles.""" v_path = Path(video_path) if not v_path.exists(): return None video_stem = v_path.stem vtd = self.temp_dir / video_stem vtd.mkdir(parents=True, exist_ok=True) start_time = time.time() calc_region = self.calculate_region( v_path, region_preset, custom_y_pct, custom_h_pct, custom_x_pct, custom_w_pct ) x, y, w, h = calc_region # Write edited SRT translated_srt = vtd / "translated.srt" srt_lines = [] for item in subtitles_data: srt_lines.append(str(item["id"])) srt_lines.append(item["timing"]) vi_text = self.normalizer.normalize(item["vietnamese_text"]) if hasattr(self.normalizer, "normalize") else item["vietnamese_text"] srt_lines.append(vi_text) srt_lines.append("") translated_srt.write_text("\n".join(srt_lines), encoding="utf-8") # 4. Cloud TTS self.progress_fn(65, "STAGE_TTS_DUBBING") self._log(f"🎙️ Tạo giọng đọc lồng tiếng ({voice}, speed={speed}x, volume={volume}%)...") dubbing_wav = vtd / "dubbing.wav" ok_tts = self.tts_engine.synthesize_srt_to_audio( str(translated_srt), str(dubbing_wav), voice=voice, speed=speed, pitch=pitch, volume=volume, temp_dir=str(vtd / "tts_segments") ) if not ok_tts or not dubbing_wav.exists(): raise RuntimeError("Lỗi tạo giọng đọc TTS.") # 5. SFX & Audio Ducking self.progress_fn(80, "STAGE_D_AUDIO_MIX") extracted_audio = vtd / "extracted_audio.wav" if not extracted_audio.exists(): cmd = [ str(self.ffmpeg_path), "-y", "-i", str(v_path), "-vn", "-acodec", "pcm_s16le", "-ar", "16000", "-ac", "1", str(extracted_audio) ] subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True) mixed_audio = vtd / "mixed_final.wav" video_dur_sec = self._get_video_duration_sec(v_path) if video_dur_sec > 0: self._log(f"⏱️ Video gốc dài {video_dur_sec:.1f}s — audio mix sẽ được pad đúng bằng, chống cắt ngọn bởi -shortest.") if mute_original_audio: self._log("🔇 Tắt hoàn toàn tiếng gốc — chỉ giữ giọng Việt (mute_original_audio=ON)") # Dubbing wav đã được time-stretch canvas đúng duration, chỉ cần chuẩn hóa sample-rate cmd_mix = [ str(self.ffmpeg_path), "-y", "-i", str(dubbing_wav), "-ac", "2", "-ar", "48000", str(mixed_audio) ] subprocess.run(cmd_mix, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True) # FIX 2p->20s: pad im lặng tới đúng video duration để -shortest không cắt video if video_dur_sec > 0: self._pad_audio_to_video_duration(mixed_audio, video_dur_sec) else: self._log("🎚️ Trộn nhạc nền (Auto Ducking) + Giọng đọc thuyết minh...") filter_complex = ( f"[0:a]volume={ducking_ratio}[bg];" f"[1:a]volume=1.0[dub];" f"[bg][dub]amix=inputs=2:duration=longest:dropout_transition=2:normalize=0[aout]" ) cmd_mix = [ str(self.ffmpeg_path), "-y", "-i", str(extracted_audio), "-i", str(dubbing_wav), "-filter_complex", filter_complex, "-map", "[aout]", "-ac", "2", "-ar", "48000", str(mixed_audio) ] subprocess.run(cmd_mix, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True) # Safety: amix duration=longest lẽ ra đã full, nhưng pad lại cho chắc if video_dur_sec > 0: self._pad_audio_to_video_duration(mixed_audio, video_dur_sec) # 6. Render Video self.progress_fn(90, "STAGE_D_RENDER") self._log("🎬 Render video hoàn thiện: Xóa sub cũ & Đè sub tiếng Việt...") final_output = self.output_dir / f"studio_final_{v_path.name}" # Probe dimensions with cv2 then ffprobe fallback (handles vertical & cv2-missing envs) vw, vh = 1920, 1080 try: cap = cv2.VideoCapture(str(v_path)) w_tmp = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH)) or 0 h_tmp = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT)) or 0 cap.release() if w_tmp > 0 and h_tmp > 0: vw, vh = w_tmp, h_tmp else: raise ValueError("cv2 returned 0") except Exception: try: import shutil as _sh _ffprobe = _sh.which("ffprobe") or str(Path(self.ffmpeg_path).parent / "ffprobe.exe") if Path(self.ffmpeg_path).exists() else "ffprobe" if not Path(_ffprobe).exists(): _ffprobe = "ffprobe" _res = subprocess.run([_ffprobe, "-v", "error", "-select_streams", "v:0", "-show_entries", "stream=width,height", "-of", "json", str(v_path)], capture_output=True, text=True, timeout=5) _j = json.loads(_res.stdout or "{}") _ws = _j.get("streams", [{}])[0] if _ws.get("width") and _ws.get("height"): vw, vh = int(_ws["width"]), int(_ws["height"]) except Exception: pass translated_ass = vtd / "translated.ass" self._convert_srt_to_ass(translated_srt, translated_ass, vw, vh, x, y, w, h, sub_style) ass_escaped = str(translated_ass).replace("\\", "/").replace(":", "\\:") sub_filter = f"subtitles='{ass_escaped}'" vf_filters = [] if w > 0 and h > 0: if sub_mask_mode == "delogo": safe_x = max(2, x) safe_y = max(2, y) safe_w = max(4, min(w, vw - safe_x - 2)) safe_h = max(4, min(h, vh - safe_y - 2)) vf_filters.append(f"delogo=x={safe_x}:y={safe_y}:w={safe_w}:h={safe_h}") elif sub_mask_mode == "box": vf_filters.append(f"drawbox=x={x}:y={y}:w={w}:h={h}:color=black@0.92:t=fill") # Determine logo & CTA overlay enabled (with fallback to bundled assets) _logo_enabled = bool(logo_enabled and logo_path and Path(logo_path).exists()) if logo_enabled and not _logo_enabled: _fb = self.base_dir / "assets" / "logo_ins_drippy4.png" if _fb.exists(): _logo_enabled = True logo_path = str(_fb) _logo_path = Path(logo_path) if _logo_enabled else None _cta_enabled = bool(cta_enabled and cta_path and Path(cta_path).exists()) if cta_enabled and not _cta_enabled: _fb2 = self.base_dir / "assets" / "cta_ins_drippy4.mp4" if _fb2.exists(): _cta_enabled = True cta_path = str(_fb2) _cta_path = Path(cta_path) if _cta_enabled else None # If any overlay enabled, we need filter_complex if _logo_enabled or _cta_enabled: if _logo_enabled: self._log(f"🖼️ Overlay logo: {Path(logo_path).name} preset={logo_preset} scale={logo_scale}% opacity={logo_opacity} chromakey={logo_chromakey}") if _cta_enabled: self._log(f"🎬 Overlay CTA: {Path(cta_path).name} preset={cta_preset} scale={cta_scale}% every {cta_interval}s for {cta_duration}s chromakey={cta_chromakey}") # Build filter parts step-by-step. Inputs: 0:video, 1:audio, 2:logo(if), 3:cta(if) # Determine indices _logo_idx = 2 if _logo_enabled else None _cta_idx = None if _cta_enabled: _cta_idx = 3 if _logo_enabled else 2 # Video preprocessing (delogo/box) -> [base] _filter_parts = [] if vf_filters and len([f for f in vf_filters if f != sub_filter]) > 0: _pre_vf = ",".join([f for f in vf_filters if f != sub_filter]) _filter_parts.append(f"[0:v]{_pre_vf}[base]") _cur = "base" else: _filter_parts.append("[0:v]null[base]") _cur = "base" # Logo overlay if _logo_enabled: _logo_w_orig, _logo_h_orig = 2400, 1792 try: _t = cv2.imread(str(_logo_path), cv2.IMREAD_UNCHANGED) if _t is not None: _logo_h_orig, _logo_w_orig = _t.shape[:2] except Exception: pass _target_w = vw * float(logo_scale) / 100.0 _sf = max(0.02, min(0.5, _target_w / float(max(1, _logo_w_orig)))) _logo_vf_parts = [] if logo_chromakey: _logo_vf_parts.append("colorkey=0x00FF00:0.3:0.1") _logo_vf_parts.append("format=rgba") _logo_vf_parts.append(f"scale=iw*{_sf:.4f}:ih*{_sf:.4f}:flags=lanczos") if float(logo_opacity) < 0.99: _logo_vf_parts.append(f"colorchannelmixer=aa={float(logo_opacity):.2f}") _logo_vf = ",".join(_logo_vf_parts) _m = 2.0 if logo_preset == "top_left": _lx, _ly = f"W*{_m/100:.3f}", f"H*{_m/100:.3f}" elif logo_preset == "top_right": _lx, _ly = f"W-w-W*{_m/100:.3f}", f"H*{_m/100:.3f}" elif logo_preset == "bottom_left": _lx, _ly = f"W*{_m/100:.3f}", f"H-h-H*{_m/100:.3f}" elif logo_preset == "bottom_right": _lx, _ly = f"W-w-W*{_m/100:.3f}", f"H-h-H*{_m/100:.3f}" elif logo_preset == "center": _lx, _ly = "(W-w)/2", "(H-h)/2" elif logo_preset == "top_center": _lx, _ly = "(W-w)/2", f"H*{_m/100:.3f}" elif logo_preset == "bottom_center": _lx, _ly = "(W-w)/2", f"H-h-H*{_m/100:.3f}" elif logo_preset == "custom": _lx, _ly = f"W*{float(logo_x_pct)/100:.4f}", f"H*{float(logo_y_pct)/100:.4f}" else: _lx, _ly = f"W-w-W*{_m/100:.3f}", f"H-h-H*{_m/100:.3f}" _filter_parts.append(f"[{_logo_idx}:v]{_logo_vf}[logo]") _next = "with_logo" if _cta_enabled or sub_filter else "v" _filter_parts.append(f"[{_cur}][logo]overlay={_lx}:{_ly}:format=rgb[{_next}]") _cur = _next # CTA overlay (periodic every interval) if _cta_enabled: # Probe CTA size _cta_w_orig, _cta_h_orig = 1080, 1920 try: _cap = cv2.VideoCapture(str(_cta_path)) _cta_w_orig = int(_cap.get(cv2.CAP_PROP_FRAME_WIDTH)) or _cta_w_orig _cta_h_orig = int(_cap.get(cv2.CAP_PROP_FRAME_HEIGHT)) or _cta_h_orig _cap.release() except Exception: pass # TikTok CTA bump: enforce minimum 42% width for mobile visibility effective_cta_scale = max(float(cta_scale), 42.0) _cta_target_w = vw * effective_cta_scale / 100.0 _cta_sf = max(0.05, min(1.0, _cta_target_w / float(max(1, _cta_w_orig)))) # Clamp height to 80% of video height to avoid overflow (CTA vertical on horizontal video) _cta_target_h = _cta_h_orig * _cta_sf _max_h = vh * 0.90 if _cta_target_h > _max_h: _cta_sf = _max_h / float(max(1, _cta_h_orig)) _cta_target_w = _cta_w_orig * _cta_sf _cta_vf_parts = [] if cta_chromakey: _cta_vf_parts.append("colorkey=0x00FF00:0.3:0.1") _cta_vf_parts.append("format=rgba") _cta_vf_parts.append(f"scale=iw*{_cta_sf:.4f}:ih*{_cta_sf:.4f}:flags=lanczos") if float(cta_opacity) < 0.99: _cta_vf_parts.append(f"colorchannelmixer=aa={float(cta_opacity):.2f}") _cta_vf = ",".join(_cta_vf_parts) _m2 = 2.0 if cta_preset == "top_left": _cx, _cy = f"W*{_m2/100:.3f}", f"H*{_m2/100:.3f}" elif cta_preset == "top_right": _cx, _cy = f"W-w-W*{_m2/100:.3f}", f"H*{_m2/100:.3f}" elif cta_preset == "bottom_left": _cx, _cy = f"W*{_m2/100:.3f}", f"H-h-H*{_m2/100:.3f}" elif cta_preset == "bottom_right": _cx, _cy = f"W-w-W*{_m2/100:.3f}", f"H-h-H*{_m2/100:.3f}" elif cta_preset == "center": _cx, _cy = "(W-w)/2", "(H-h)/2" elif cta_preset == "top_center": _cx, _cy = "(W-w)/2", f"H*{_m2/100:.3f}" elif cta_preset == "bottom_center": _cx, _cy = "(W-w)/2", f"H-h-H*{_m2/100:.3f}" elif cta_preset == "custom": _cx, _cy = f"W*{float(cta_x_pct)/100:.4f}", f"H*{float(cta_y_pct)/100:.4f}" else: _cx, _cy = "(W-w)/2", f"H-h-H*{_m2/100:.3f}" _enable = f"lt(mod(t\\,{float(cta_interval)})\\,{float(cta_duration)})" _filter_parts.append(f"[{_cta_idx}:v]{_cta_vf}[cta]") _next2 = "v" if not sub_filter else "with_cta" # Use escaped comma for FFmpeg enable expression _filter_parts.append(f"[{_cur}][cta]overlay={_cx}:{_cy}:format=rgb:enable='{_enable}'[{_next2}]") _cur = _next2 # TikTok polish: CFR 30 + even dims + light sharpen BEFORE subtitles (keeps text razor sharp) tiktok_polish = "fps=30:round=near,scale=trunc(iw/2)*2:trunc(ih/2)*2:flags=lanczos+accurate_rnd+full_chroma_int:sws_dither=ed,unsharp=3:3:0.35:3:3:0.0" if sub_filter: _filter_parts.append(f"[{_cur}]{tiktok_polish}[polished]") _cur = "polished" _filter_parts.append(f"[{_cur}]{sub_filter}[v]") _cur = "v" else: _filter_parts.append(f"[{_cur}]{tiktok_polish}[v]") _cur = "v" _filter_complex = ";".join(_filter_parts) # Build ffmpeg inputs: video, audio, logo(if), cta(if) - add -shortest when looped overlays present cmd_inputs = [str(self.ffmpeg_path), "-y", "-fflags", "+genpts", "-avoid_negative_ts", "make_zero", "-i", str(v_path), "-i", str(mixed_audio)] if _logo_enabled: cmd_inputs += ["-loop", "1", "-i", str(_logo_path)] if _cta_enabled: cmd_inputs += ["-stream_loop", "999", "-i", str(_cta_path)] _extra = ["-shortest"] if (_logo_enabled or _cta_enabled) else [] # TikTok spec: High Profile yuv420p, 30 CFR, 48k AAC, faststart, accurate sync tiktok_video_args = ["-r", "30", "-c:v", "libx264", "-preset", "medium", "-crf", "18", "-profile:v", "high", "-level", "4.1", "-pix_fmt", "yuv420p", "-g", "60", "-keyint_min", "30", "-sc_threshold", "0", "-x264-params", "ref=4:bframes=2:me=hex:subme=7:psy=1:psy-rd=0.8:aq-mode=2", "-colorspace", "bt709", "-color_primaries", "bt709", "-color_trc", "bt709", "-color_range", "tv"] tiktok_audio_args = ["-c:a", "aac", "-profile:a", "aac_low", "-ar", "48000", "-ac", "2", "-b:a", "192k", "-af", "aresample=async=1:min_hard_comp=0.100000:first_pts=0"] tiktok_mux_args = ["-movflags", "+faststart", "-fflags", "+genpts", "-max_interleave_delta", "100M", "-vsync", "cfr", "-fps_mode", "cfr"] cmd_render = cmd_inputs + ["-filter_complex", _filter_complex, "-map", "[v]", "-map", "1:a:0"] + _extra + tiktok_video_args + tiktok_audio_args + tiktok_mux_args + ["-shortest", str(final_output)] res = subprocess.run(cmd_render, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True) if res.returncode != 0: self._log(f"⚠️ Overlay render failed ({res.stderr[:300]}), fallback to TikTok-spec normal render...") # Rebuild VF with TikTok polish inserted before subs (same as non-overlay branch) tiktok_polish_fb = "fps=30:round=near,scale=trunc(iw/2)*2:trunc(ih/2)*2:flags=lanczos+accurate_rnd+full_chroma_int:sws_dither=ed,unsharp=3:3:0.35:3:3:0.0" vf_without_sub_fb = [f for f in vf_filters if f != sub_filter] ordered_fb = [] if vf_without_sub_fb: ordered_fb.extend(vf_without_sub_fb) ordered_fb.append(tiktok_polish_fb) ordered_fb.append(sub_filter) final_vf = ",".join(ordered_fb) cmd_render = [ str(self.ffmpeg_path), "-y", "-fflags", "+genpts", "-avoid_negative_ts", "make_zero", "-i", str(v_path), "-i", str(mixed_audio), "-vf", final_vf, "-map", "0:v:0", "-map", "1:a:0", "-r", "30", "-c:v", "libx264", "-preset", "medium", "-crf", "18", "-profile:v", "high", "-level", "4.1", "-pix_fmt", "yuv420p", "-g", "60", "-keyint_min", "30", "-sc_threshold", "0", "-x264-params", "ref=4:bframes=2:me=hex:subme=7:psy=1:psy-rd=0.8:aq-mode=2", "-colorspace", "bt709", "-color_primaries", "bt709", "-color_trc", "bt709", "-color_range", "tv", "-c:a", "aac", "-profile:a", "aac_low", "-ar", "48000", "-ac", "2", "-b:a", "192k", "-af", "aresample=async=1:min_hard_comp=0.100000:first_pts=0", "-movflags", "+faststart", "-fflags", "+genpts", "-max_interleave_delta", "100M", "-vsync", "cfr", "-fps_mode", "cfr", "-shortest", str(final_output) ] res = subprocess.run(cmd_render, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True) if res.returncode != 0: self._log(f"⚠️ Fallback render không mask: {res.stderr[:120]}") fallback_cmd = [ str(self.ffmpeg_path), "-y", "-fflags", "+genpts", "-i", str(v_path), "-i", str(mixed_audio), "-vf", f"{tiktok_polish_fb},{sub_filter}", "-map", "0:v:0", "-map", "1:a:0", "-r", "30", "-c:v", "libx264", "-preset", "medium", "-crf", "18", "-profile:v", "high", "-pix_fmt", "yuv420p", "-c:a", "aac", "-profile:a", "aac_low", "-ar", "48000", "-ac", "2", "-b:a", "192k", "-movflags", "+faststart", "-shortest", str(final_output) ] subprocess.run(fallback_cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True) else: # TikTok polish inserted before subtitles for sharpness + CFR + even dims tiktok_polish = "fps=30:round=near,scale=trunc(iw/2)*2:trunc(ih/2)*2:flags=lanczos+accurate_rnd+full_chroma_int:sws_dither=ed,unsharp=3:3:0.35:3:3:0.0" # Order: mask (delogo/box) -> polish -> subtitles vf_without_sub = [f for f in vf_filters if f != sub_filter] # Rebuild vf in TikTok-optimal order ordered_vf = [] if vf_without_sub: ordered_vf.extend(vf_without_sub) ordered_vf.append(tiktok_polish) ordered_vf.append(sub_filter) final_vf = ",".join(ordered_vf) cmd_render = [ str(self.ffmpeg_path), "-y", "-fflags", "+genpts", "-avoid_negative_ts", "make_zero", "-i", str(v_path), "-i", str(mixed_audio), "-vf", final_vf, "-map", "0:v:0", "-map", "1:a:0", "-r", "30", "-c:v", "libx264", "-preset", "medium", "-crf", "18", "-profile:v", "high", "-level", "4.1", "-pix_fmt", "yuv420p", "-g", "60", "-keyint_min", "30", "-sc_threshold", "0", "-x264-params", "ref=4:bframes=2:me=hex:subme=7:psy=1:psy-rd=0.8:aq-mode=2", "-colorspace", "bt709", "-color_primaries", "bt709", "-color_trc", "bt709", "-color_range", "tv", "-c:a", "aac", "-profile:a", "aac_low", "-ar", "48000", "-ac", "2", "-b:a", "192k", "-af", "aresample=async=1:min_hard_comp=0.100000:first_pts=0", "-movflags", "+faststart", "-fflags", "+genpts", "-max_interleave_delta", "100M", "-vsync", "cfr", "-fps_mode", "cfr", "-shortest", str(final_output) ] res = subprocess.run(cmd_render, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True) if res.returncode != 0: self._log(f"⚠️ Fallback render không mask: {res.stderr[:120]} | retry with minimal TikTok spec") # Minimal fallback: at least ensure TikTok audio/video spec fallback_cmd = [ str(self.ffmpeg_path), "-y", "-fflags", "+genpts", "-i", str(v_path), "-i", str(mixed_audio), "-vf", f"{tiktok_polish},{sub_filter}", "-map", "0:v:0", "-map", "1:a:0", "-r", "30", "-c:v", "libx264", "-preset", "medium", "-crf", "18", "-profile:v", "high", "-pix_fmt", "yuv420p", "-c:a", "aac", "-profile:a", "aac_low", "-ar", "48000", "-ac", "2", "-b:a", "192k", "-movflags", "+faststart", "-shortest", str(final_output) ] subprocess.run(fallback_cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True) total_elapsed = time.time() - start_time # Verify output duration vs input — báo động ngay nếu bị cắt ngọn (2p->20s) try: out_dur = self._get_video_duration_sec(final_output) in_dur = video_dur_sec or self._get_video_duration_sec(v_path) if in_dur > 0 and out_dur > 0: if out_dur < in_dur - 3.0: self._log(f"❌ CẢNH BÁO ĐỘ DÀI: input {in_dur:.1f}s nhưng output chỉ {out_dur:.1f}s " f"(thiếu {in_dur - out_dur:.1f}s)! Kiểm tra phụ đề/ASR-OCR có bị rụng đuôi không.") else: self._log(f"⏱️ Verify độ dài OK: input {in_dur:.1f}s -> output {out_dur:.1f}s.") except Exception: pass self.progress_fn(100, "DONE") self._log(f"🎉 HOÀN THÀNH XUẤT SẮC TRONG {total_elapsed:.1f}s!") self._log(f"📁 Video đã lưu tại: {final_output}") return str(final_output) def run_video( self, video_path: str, mode: str = "asr", source_lang: str = "zh", voice: str = "vi-VN-NamMinhNeural", speed: float = 1.0, pitch: int = 0, volume: int = 100, region_preset: str = "custom", sub_mask_mode: str = "box", sub_style: str = "motion_drip", custom_y_pct: float = 75.0, custom_h_pct: float = 20.0, custom_x_pct: float = 0.0, custom_w_pct: float = 100.0, ducking_ratio: float = 0.18, enable_sfx: bool = True, mute_original_audio: bool = False, logo_enabled: bool = False, logo_path: str = "", logo_preset: str = "bottom_right", logo_scale: float = 15.0, logo_opacity: float = 1.0, logo_x_pct: float = 80.0, logo_y_pct: float = 80.0, logo_chromakey: bool = True, cta_enabled: bool = False, cta_path: str = "", cta_preset: str = "bottom_center", cta_scale: float = 35.0, cta_opacity: float = 1.0, cta_x_pct: float = 50.0, cta_y_pct: float = 85.0, cta_interval: float = 5.0, cta_duration: float = 2.0, cta_chromakey: bool = True ) -> Optional[str]: """Full 1-Click End-to-End Automatic Dubbing Pipeline.""" v_path = Path(video_path) if not v_path.exists(): self._log(f"❌ Video not found: {video_path}") return None # Register into SQLite Job DB try: self.job_manager.register_job(str(v_path)) except Exception: pass try: # 1. Extract subtitles subs_res = self.extract_subtitles_only( video_path=video_path, mode=mode, source_lang=source_lang, region_preset=region_preset, custom_y_pct=custom_y_pct, custom_h_pct=custom_h_pct, custom_x_pct=custom_x_pct, custom_w_pct=custom_w_pct ) if not subs_res: return None # 2. Render from subtitles return self.render_from_subtitles( video_path=video_path, subtitles_data=subs_res["blocks"], voice=voice, speed=speed, pitch=pitch, volume=volume, region_preset=region_preset, sub_mask_mode=sub_mask_mode, sub_style=sub_style, custom_y_pct=custom_y_pct, custom_h_pct=custom_h_pct, custom_x_pct=custom_x_pct, custom_w_pct=custom_w_pct, ducking_ratio=ducking_ratio, enable_sfx=enable_sfx, mute_original_audio=mute_original_audio, logo_enabled=logo_enabled, logo_path=logo_path, logo_preset=logo_preset, logo_scale=logo_scale, logo_opacity=logo_opacity, logo_x_pct=logo_x_pct, logo_y_pct=logo_y_pct, logo_chromakey=logo_chromakey, cta_enabled=cta_enabled, cta_path=cta_path, cta_preset=cta_preset, cta_scale=cta_scale, cta_opacity=cta_opacity, cta_x_pct=cta_x_pct, cta_y_pct=cta_y_pct, cta_interval=cta_interval, cta_duration=cta_duration, cta_chromakey=cta_chromakey ) except Exception as e: self._log(f"❌ LỖI PIPELINE: {str(e)}") self.progress_fn(0, "FAILED") return None def _convert_srt_to_ass( self, srt_path: Path, ass_path: Path, vw: int, vh: int, box_x: int, box_y: int, box_w: int, box_h: int, sub_style: str = "motion_drip" ): """Converts SRT to styled ASS format for TikTok/Shorts typography.""" content = srt_path.read_text(encoding="utf-8", errors="ignore") # TikTok safe zone: keep subtitle inside 82% height, 7% bottom margin, larger font for mobile legibility safe_margin_v = max(42, int(vh * 0.07)) raw_margin_v = int(vh - (box_y + box_h * 0.75)) if box_h > 0 else safe_margin_v margin_v = max(safe_margin_v, raw_margin_v) # Clamp to avoid bottom UI overlap max_bottom = int(vh * 0.12) if margin_v < max_bottom: margin_v = max_bottom # Larger font for TikTok: vertical 56-72, horizontal 48-64 if box_h > 0: base = int(box_h * 0.52) if vh > vw: # vertical TikTok font_size = max(42, min(72, base)) else: font_size = max(38, min(64, base)) else: font_size = 58 if vh > vw else 50 # Typography Color & Outline palettes - TikTok optimized for high compression if sub_style == "motion_drip" or sub_style == "yellow_bold": font_color = "&H0000FFFF" # Bright Yellow outline_color = "&H00000000" # Pure Black back_color = "&H99000000" outline = 5.0 shadow = 2.2 elif sub_style == "neon_cyan": font_color = "&H00FFFF00" # Cyan outline_color = "&H00000000" back_color = "&H99000000" outline = 5.0 shadow = 2.2 elif sub_style == "capsule_tag": font_color = "&H00FFFFFF" # White outline_color = "&H00111111" back_color = "&HBB000000" outline = 3.2 shadow = 0.0 else: # white_bold font_color = "&H00FFFFFF" # Crisp White outline_color = "&H00000000" back_color = "&H99000000" outline = 4.8 shadow = 2.0 ass_header = f"""[Script Info] ScriptType: v4.00+ PlayResX: {vw} PlayResY: {vh} ScaledBorderAndShadow: yes [V4+ Styles] Format: Name, Fontname, Fontsize, PrimaryColour, SecondaryColour, OutlineColour, BackColour, Bold, Italic, Underline, StrikeOut, ScaleX, ScaleY, Spacing, Angle, BorderStyle, Outline, Shadow, Alignment, MarginL, MarginR, MarginV, Encoding Style: Default,DejaVu Sans,{font_size},{font_color},&H000000FF,{outline_color},{back_color},1,0,0,0,100,100,0,0,1,{outline},{shadow},2,40,40,{margin_v},1 [Events] Format: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text """ events = [] pattern = r"(\d+)\s+(\d{2}:\d{2}:\d{2}[.,]\d{3})\s*-->\s*(\d{2}:\d{2}:\d{2}[.,]\d{3})\s*\n(.*?)(?=\n\s*\d+\s+\d{2}:\d{2}:\d{2}[.,]\d{3}\s*-->|\Z)" for m in re.finditer(pattern, content, re.DOTALL): start_str = m.group(2).strip().replace(",", ".") end_str = m.group(3).strip().replace(",", ".") s_parts = start_str.split(":") e_parts = end_str.split(":") s_ass = f"{int(s_parts[0])}:{s_parts[1]}:{float(s_parts[2]):05.2f}" e_ass = f"{int(e_parts[0])}:{e_parts[1]}:{float(e_parts[2]):05.2f}" text = " ".join(line.strip() for line in m.group(4).splitlines() if line.strip()) if text: # FIX: escape ký tự ASS đặc biệt — "{deal}", "\", emoticon bị libass # hiểu thành override-tag -> render lỗi/hiện chữ thô text = text.replace("\\", "\\\\").replace("{", "\\{").replace("}", "\\}") events.append(f"Dialogue: 0,{s_ass},{e_ass},Default,,0,0,0,,{text}") ass_path.write_text(ass_header + "\n".join(events), encoding="utf-8") def _parse_srt_blocks(self, content: str) -> List[Dict]: # FIX: giờ 1 chữ số (\d{1,2}) — timestamp "0:00:01,000" của Whisper/tool lẻ bị miss cả block pattern = r"(\d+)\s+(\d{1,2}:\d{2}:\d{2}[.,]\d{3})\s*-->\s*(\d{1,2}:\d{2}:\d{2}[.,]\d{3})\s*\n(.*?)(?=\n\s*\d+\s+\d{1,2}:\d{2}:\d{2}[.,]\d{3}\s*-->|\Z)" blocks = [] for m in re.finditer(pattern, content, re.DOTALL): text = " ".join(line.strip() for line in m.group(4).splitlines() if line.strip()) start_str = m.group(2).strip() end_str = m.group(3).strip() if text: blocks.append({ "id": int(m.group(1)), "timing": f"{start_str} --> {end_str}", "text": text, "start_ms": self._ts_to_ms(start_str), "end_ms": self._ts_to_ms(end_str), }) return blocks @staticmethod def _ts_to_ms(ts: str) -> int: ts = ts.strip().replace(".", ",") m = re.match(r"(\d+):(\d+):(\d+)[,](\d+)", ts) if m: h, mins, s, ms = map(int, m.groups()) return ((h * 3600 + mins * 60 + s) * 1000) + ms return 0 def _build_system_prompt(self) -> str: glossary_sample = ", ".join([f"{k} -> {v}" for k, v in list(self.glossary.items())[:35]]) return ( "Bạn là chuyên gia dịch thuật video Sneaker, Thời trang Streetwear, Review sản phẩm từ tiếng Trung/Anh sang tiếng Việt tự nhiên, sành điệu, bắt trend Gen Z.\n" "QUY TẮC BẮT BUỘC:\n" "1. Dịch từng dòng theo cấu trúc: [N] Câu dịch tiếng Việt hoàn chỉnh.\n" "2. Giữ nguyên thuật ngữ & thương hiệu tiếng Anh (Nike, Jordan, Yeezy, BAPE, Supreme, Rick Owens, outfit, fit, drip, full box, collab, signature...). \n" f"3. Bắt buộc áp dụng từ điển chuyên ngành: {glossary_sample}\n" "4. Dịch ĐẦY ĐỦ Ý NGHĨA trọn vẹn của câu, giữ đủ các từ khoá, TUYỆT ĐỐI KHÔNG cắt xén, không bỏ lửng hay cắt cụt mất từ ở đầu, giữa hoặc cuối câu. Câu dịch phải trọn vẹn ngữ nghĩa, tự nhiên và dễ hiểu.\n" "5. KHÔNG giải thích, CHỈ trả về danh sách các dòng [N] Tiếng Việt." ) def _translate_srt_cloud(self, srt_in: Path, srt_out: Path, source_lang: str = "zh"): content = srt_in.read_text(encoding="utf-8", errors="ignore") blocks = self._parse_srt_blocks(content) if not blocks: srt_out.write_text(content, encoding="utf-8") return system_prompt = self._build_system_prompt() validator = TranslationValidator() trans_map = {} # Phase 1: Batch translation in chunks of 20 chunk_size = 20 for i in range(0, len(blocks), chunk_size): chunk = blocks[i : i + chunk_size] res = self._translate_chunk_with_retry(chunk, system_prompt, validator, max_retries=2) trans_map.update(res) # Phase 2: Retry missing / Chinese-leaked blocks missing_blocks = [b for b in blocks if (b["id"] not in trans_map or has_chinese(trans_map.get(b["id"], "")))] if missing_blocks: self._log(f"🔄 Đang hoàn thiện nốt {len(missing_blocks)} câu dịch còn lại...") retry_result = self._translate_chunk_with_retry(missing_blocks, system_prompt, validator, max_retries=2) trans_map.update(retry_result) # Phase 3: Google Translate fallback still_missing = [b for b in blocks if (b["id"] not in trans_map or has_chinese(trans_map.get(b["id"], "")))] if still_missing: self._log(f"🌐 {len(still_missing)} câu cần fallback Google Translate...") google_fb = GoogleTranslateFallback(log_fn=self._log) google_result = google_fb.translate_blocks(still_missing) for bid_str, text in google_result.items(): trans_map[int(bid_str)] = text # Phase 4: Post-editing self._log("✨ Chạy Post-Editor sửa lỗi dịch...") for b in blocks: bid = b["id"] if bid in trans_map: trans_map[bid] = post_edit_translation(b["text"], trans_map[bid]) # Phase 5: Quality Guard self._log("🛡️ Quality Guard kiểm tra chất lượng bản dịch...") guard = TranslationQualityGuard(min_score=70) guard_dict = {str(b["id"]): trans_map.get(b["id"], b["text"]) for b in blocks} fixed_dict, quality_report = guard.audit_and_fix(blocks, guard_dict) for b in blocks: bid_str = str(b["id"]) if bid_str in fixed_dict: trans_map[b["id"]] = fixed_dict[bid_str] # Phase 6: Subtitle text cleanup (safe formatting) for b in blocks: bid = b["id"] if bid in trans_map: t = str(trans_map[bid]).strip() t = re.sub(r"\s+", " ", t).strip(" ,") trans_map[bid] = t # Phase 7: Output SRT out_lines = [] for b in blocks: vi_text = trans_map.get(b["id"], b["text"]) out_lines.append(str(b["id"])) out_lines.append(b["timing"]) out_lines.append(vi_text) out_lines.append("") srt_out.write_text("\n".join(out_lines), encoding="utf-8") self._log(f"✅ Dịch hoàn tất {len(blocks)} câu với 7 bước kiểm tra chất lượng.") def _translate_chunk_with_retry( self, chunk: List[Dict], system_prompt: str, validator: TranslationValidator, max_retries: int = 2 ) -> Dict[int, str]: prompt_lines = [f"[{b['id']}] {b['text']}" for b in chunk] full_transcript = "\n".join(prompt_lines) result = {} for attempt in range(max_retries + 1): raw_res = self._direct_25_model_translate(system_prompt, full_transcript) cleaned_res = re.sub(r".*?", "", raw_res, flags=re.DOTALL).strip() attempt_map = {} for line in cleaned_res.splitlines(): m = re.match(r"^\s*\[(\d+)\]\s*(.*)$", line.strip()) if m: attempt_map[int(m.group(1))] = m.group(2).strip() # FIX rụng đuôi dịch: chỉ accept early-return khi ĐỦ 100% số dòng chunk. # Trước đây subset 5/20 dòng pass validate -> return thiếu 15 dòng im lặng. expected_ids = {b["id"] for b in chunk} got_ids = set(attempt_map.keys()) & expected_ids val_dict = {str(b["id"]): attempt_map.get(b["id"], "") for b in chunk if b["id"] in attempt_map} val_chunk = [b for b in chunk if b["id"] in attempt_map] if val_chunk and val_dict: is_valid, error_msg = validator.validate(val_chunk, val_dict) if is_valid and got_ids == expected_ids: result.update(attempt_map) return result if not is_valid: self._log(f"⚠️ Chunk validate fail ({len(got_ids)}/{len(expected_ids)} dòng): {error_msg} — retry...") elif got_ids != expected_ids: missing = sorted(expected_ids - got_ids) self._log(f"⚠️ Chunk thiếu {len(missing)}/{len(expected_ids)} dòng {missing[:8]} — retry...") for bid, text in attempt_map.items(): src_block = next((b for b in chunk if b["id"] == bid), None) if src_block: ok, _ = validator.validate_single_block(bid, src_block["text"], text) if ok: result[bid] = text elif attempt_map: # Không parse được dòng nào theo format [N] — giữ lại để Phase 2/3 xử lý self._log(f"⚠️ Chunk parse được 0/{len(expected_ids)} dòng hợp lệ — retry...") return result NIM_TRANSLATION_MODEL = "nvidia/nemotron-3.5-lightning-30b-a3b" NIM_TRANSLATION_URL = "https://integrate.api.nvidia.com/v1/chat/completions" def _get_nim_keys(self): """Lấy Nvidia NIM keys từ .env/env: NIM_API_KEY > NVIDIA_KEY_1..N.""" import os keys = [] direct = (os.getenv("NIM_API_KEY", "") or "").strip() if direct: keys.append(direct) # Ưu tiên engine đã load sẵn (.env + os.environ) for src in (getattr(self.asr_engine, "nvidia_keys", []), getattr(self.ocr_engine, "nvidia_keys", [])): for k in (src or []): if k and k not in keys: keys.append(k) # Quét NVIDIA_KEY_N trực tiếp từ environ (cho HF Secrets) for i in range(1, 20): v = (os.getenv(f"NVIDIA_KEY_{i}", "") or "").strip() if v and v not in keys: keys.append(v) return keys def _direct_25_model_translate(self, system_prompt: str, user_content: str) -> str: messages = [ {"role": "system", "content": system_prompt}, {"role": "user", "content": user_content} ] # 0. NVIDIA NIM Nemotron 3.5 Lightning 30B (ƯU TIÊN #1 — key có sẵn trong tool) nim_keys = self._get_nim_keys() for key in nim_keys: try: res = requests.post( self.NIM_TRANSLATION_URL, headers={"Authorization": f"Bearer {key}", "Content-Type": "application/json"}, json={"model": self.NIM_TRANSLATION_MODEL, "messages": messages, "temperature": 0.2, "max_tokens": 4096}, timeout=30 ) if res.status_code == 200: content = res.json()["choices"][0]["message"]["content"].strip() content = re.sub(r".*?", "", content, flags=re.DOTALL).strip() if len(content) > 15: self._log(f"✅ Dịch bằng NVIDIA NIM ({self.NIM_TRANSLATION_MODEL})") return content else: self._log(f"⚠️ NIM {res.status_code}: {res.text[:120]}") except Exception as e: self._log(f"⚠️ NIM lỗi: {e}") continue # 1. Google Gemini (Fastest & highest accuracy for Asian languages) gemini_keys = self.ocr_engine.gemini_keys for model in ["gemini-3.6-flash", "gemini-3.5-flash", "gemini-flash-latest", "gemini-pro-latest", "gemini-2.5-flash", "gemini-2.0-flash"]: for key in gemini_keys: try: url = f"https://generativelanguage.googleapis.com/v1beta/models/{model}:generateContent?key={key}" payload = { "contents": [{"parts": [{"text": f"{system_prompt}\n\n{user_content}"}]}], "generationConfig": {"temperature": 0.2, "maxOutputTokens": 4096} } res = requests.post(url, json=payload, timeout=25) if res.status_code == 200: content = res.json()["candidates"][0]["content"]["parts"][0]["text"].strip() if len(content) > 15: return content except Exception: pass # 2. Groq (High speed LLM) groq_keys = self.asr_engine.groq_keys for model in ["openai/gpt-oss-20b", "groq/compound", "qwen/qwen3.6-27b", "llama-3.3-70b-versatile", "llama-3.1-8b-instant"]: for key in groq_keys: try: res = requests.post( "https://api.groq.com/openai/v1/chat/completions", headers={"Authorization": f"Bearer {key}", "Content-Type": "application/json"}, json={"model": model, "messages": messages, "temperature": 0.2, "max_tokens": 4096}, timeout=25 ) if res.status_code == 200: content = res.json()["choices"][0]["message"]["content"].strip() content = re.sub(r".*?", "", content, flags=re.DOTALL).strip() if len(content) > 15: return content except Exception: pass # 3. OpenRouter Free & Standard Models or_keys = self.ocr_engine.openrouter_keys for model in [ "nvidia/nemotron-3.5-lightning:free", "inclusionai/ling-3.0-flash-fin:free", "liquid/lfm-2.5-2.6b:free", "deepseek/deepseek-r1:free", "qwen/qwen-2.5-72b-instruct:free" ]: for key in or_keys: try: res = requests.post( "https://openrouter.ai/api/v1/chat/completions", headers={"Authorization": f"Bearer {key}", "Content-Type": "application/json", "HTTP-Referer": "https://trungsangviet.local", "X-Title": "TrungSangViet"}, json={"model": model, "messages": messages, "temperature": 0.2, "max_tokens": 4096}, timeout=30 ) if res.status_code == 200: content = res.json()["choices"][0]["message"]["content"].strip() content = re.sub(r".*?", "", content, flags=re.DOTALL).strip() if len(content) > 15: return content except Exception: pass # FIX: trước đây fail hết model thì trả source (tiếng Trung) như "đã dịch" -> # caller parse thành trans_map, có thể lọt qua validator vào SRT. # Nay trả "" để Phase 2 retry + Phase 3 Google fallback làm việc đúng. self._log("❌ Tất cả translation models đều fail cho chunk này — để trống cho fallback.") return ""