diff --git "a/app/core/cloud_pipeline.py" "b/app/core/cloud_pipeline.py" --- "a/app/core/cloud_pipeline.py" +++ "b/app/core/cloud_pipeline.py" @@ -1,1208 +1,1208 @@ -""" -app/core/cloud_pipeline.py -────────────────────────── -Pure Cloud-Native Headless Pipeline Coordinator. -Features: -1. High-Definition Video Preview Frame Generator using FFmpeg (100% Reliable). -2. Multi-Model ASR Matrix (Groq Whisper Large V3 / OpenRouter / Gemini). -3. AI Vision Models OCR with automatic ASR fallback. -4. Active 2026 AI Translation Models (Groq Qwen 3.6 / GPT-OSS 120B / Gemini Flash / Nemotron 3). -5. Glossary & Slang Mapping (Fashion, Sneaker, Streetwear, Chinese Slang). -6. 7-Step Quality Guard & Semantic Validation & Timing Compaction. -7. Subtitle Editing Hooks (extract_subtitles_only & render_from_subtitles). -8. ASS Subtitle Engine with TikTok / Shorts Typography & Word Jump Styles. -9. 100% Solid/Delogo Masking Box to eliminate old Chinese subtitles. -10. Studio SFX & Audio Ducking Mixer. -11. SQLite JobManager Integration for state persistence. -""" - -import os -import re -import sys -import json -import time -import base64 -import shutil -import cv2 -import requests -import subprocess -from pathlib import Path -from typing import Optional, Callable, Dict, Any, Tuple, List - -from app.core.cloud_asr import CloudASREngine -from app.core.cloud_ocr import CloudOCREngine -from app.core.cloud_tts import CloudTTSEngine -from app.core.vietnamese_text_normalizer import VietnameseTextNormalizer -from app.core.translation_validator import TranslationValidator -from app.core.translation_quality_guard import TranslationQualityGuard -from app.core.translation_post_editor import post_edit_translation -from app.core.google_translate_fallback import GoogleTranslateFallback -from app.core.subtitle_compactor import compact_blocks_for_tts -from app.core.job_manager import JobManager -from app.core.studio_sfx import generate_sfx_clip - - -def has_chinese(text: str) -> bool: - """Returns True if string contains CJK Chinese characters.""" - return bool(re.search(r"[\u4e00-\u9fff]", text)) - - -class CloudPipeline: - def __init__( - self, - base_dir: Optional[Path] = None, - log_callback: Optional[Callable[[str], None]] = None, - progress_callback: Optional[Callable[[int, str], None]] = None, - ffmpeg_path: str = "ffmpeg" - ): - self.base_dir = Path(base_dir or Path(__file__).resolve().parents[2]) - self.output_dir = self.base_dir / "output" - self.temp_dir = self.base_dir / "temp" - self.output_dir.mkdir(parents=True, exist_ok=True) - self.temp_dir.mkdir(parents=True, exist_ok=True) - - self.log_fn = log_callback or print - self.progress_fn = progress_callback or (lambda pct, stage: None) - self.ffmpeg_path = ffmpeg_path - - self.asr_engine = CloudASREngine(log_fn=self._log) - self.ocr_engine = CloudOCREngine(log_fn=self._log) - self.tts_engine = CloudTTSEngine(log_fn=self._log, ffmpeg_path=self.ffmpeg_path) - self.normalizer = VietnameseTextNormalizer() - self.job_manager = JobManager.instance() - self.glossary = self._load_glossary() - - def _log(self, msg: str): - self.log_fn(f"[Cloud Pipeline] {msg}") - - def _load_glossary(self) -> Dict[str, str]: - glossary_path = self.base_dir / "glossary.json" - if glossary_path.exists(): - try: - data = json.loads(glossary_path.read_text(encoding="utf-8")) - return data.get("glossary", {}) - except Exception as e: - self._log(f"⚠️ Lỗi đọc glossary.json: {e}") - return {} - - def calculate_region( - self, - video_path: Path, - region_preset: str = "custom", - custom_y_pct: float = 75.0, - custom_h_pct: float = 20.0, - custom_x_pct: float = 0.0, - custom_w_pct: float = 100.0 - ) -> Tuple[int, int, int, int]: - vw, vh = 1920, 1080 - try: - cap = cv2.VideoCapture(str(video_path)) - w = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH)) - h = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT)) - cap.release() - if w > 0 and h > 0: - vw, vh = w, h - except Exception: - pass - - if region_preset == "bottom_25": - x = 0 - w = vw - y = int(vh * 0.72) - h = int(vh * 0.24) - elif region_preset == "bottom_15": - x = 0 - w = vw - y = int(vh * 0.82) - h = int(vh * 0.15) - elif region_preset == "middle_25": - x = 0 - w = vw - y = int(vh * 0.38) - h = int(vh * 0.24) - elif region_preset == "none": - return (0, 0, 0, 0) - elif region_preset == "custom": - x = int(vw * (custom_x_pct / 100.0)) - w = int(vw * (custom_w_pct / 100.0)) - y = int(vh * (custom_y_pct / 100.0)) - h = int(vh * (custom_h_pct / 100.0)) - else: - return (0, 0, 0, 0) - - x = max(0, min(vw - 2, x)) - y = max(0, min(vh - 2, y)) - w = max(2, min(vw - x, w)) - h = max(2, min(vh - y, h)) - return (x, y, w, h) - - def generate_preview_frame( - self, - video_path: str, - region_preset: str = "custom", - custom_y_pct: float = 75.0, - custom_h_pct: float = 20.0, - custom_x_pct: float = 0.0, - custom_w_pct: float = 100.0, - sec: float = 2.0, - return_type: str = "rgb", - # ── Logo overlay params ── - logo_enabled: bool = False, - logo_path: str = "", - logo_preset: str = "bottom_right", - logo_scale: float = 15.0, - logo_opacity: float = 1.0, - logo_x_pct: float = 80.0, - logo_y_pct: float = 80.0, - logo_chromakey: bool = True, - # ── CTA video overlay params (appears every 5s) ── - cta_enabled: bool = False, - cta_path: str = "", - cta_preset: str = "bottom_center", - cta_scale: float = 35.0, - cta_opacity: float = 1.0, - cta_x_pct: float = 50.0, - cta_y_pct: float = 85.0, - cta_interval: float = 5.0, - cta_duration: float = 2.0, - cta_chromakey: bool = True - ): - """Extracts a frame at timestamp sec and overlays bounding box + logo + CTA preview.""" - v_file = Path(video_path) - if not v_file.exists(): - return None - - raw_frame_path = self.temp_dir / f"raw_frame_{int(time.time()*1000)}.jpg" - cmd = [ - str(self.ffmpeg_path), "-ss", str(sec), "-y", - "-i", str(v_file), - "-frames:v", "1", "-q:v", "2", - str(raw_frame_path) - ] - try: - subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True) - frame = cv2.imread(str(raw_frame_path)) - if raw_frame_path.exists(): - raw_frame_path.unlink() - except Exception: - frame = None - - if frame is None: - cap = cv2.VideoCapture(str(v_file)) - fps = cap.get(cv2.CAP_PROP_FPS) or 25.0 - target_frame = max(0, int(sec * fps)) - cap.set(cv2.CAP_PROP_POS_FRAMES, target_frame) - ret, frame = cap.read() - cap.release() - - if frame is None: - return None - - x, y, w, h = self.calculate_region( - v_file, region_preset, custom_y_pct, custom_h_pct, custom_x_pct, custom_w_pct - ) - - if w > 0 and h > 0: - overlay = frame.copy() - cv2.rectangle(overlay, (x, y), (x + w, y + h), (0, 0, 255), -1) - cv2.addWeighted(overlay, 0.35, frame, 0.65, 0, frame) - cv2.rectangle(frame, (x, y), (x + w, y + h), (0, 255, 255), 3) - - label = "VUNG CHE SUB CU & QUET OCR" - font = cv2.FONT_HERSHEY_SIMPLEX - font_scale = max(0.5, frame.shape[1] / 1200.0) - thickness = 2 - (tw, th), _ = cv2.getTextSize(label, font, font_scale, thickness) - - label_y = max(th + 10, y - 8) - cv2.rectangle(frame, (x, label_y - th - 6), (x + tw + 10, label_y + 4), (0, 0, 0), -1) - cv2.putText(frame, label, (x + 5, label_y), font, font_scale, (0, 255, 255), thickness, cv2.LINE_AA) - - # ── Logo overlay preview ── - if logo_enabled and logo_path and Path(logo_path).exists(): - try: - from app.core.logo_overlay import load_logo_rgba, overlay_logo_on_frame - _logo_rgba = load_logo_rgba(logo_path, chromakey=logo_chromakey) - if _logo_rgba is not None: - frame = overlay_logo_on_frame( - frame_bgr=frame, - logo_rgba=_logo_rgba, - preset=logo_preset, - scale_pct=float(logo_scale), - opacity=float(logo_opacity), - custom_x_pct=float(logo_x_pct), - custom_y_pct=float(logo_y_pct), - margin_pct=2.0 - ) - cv2.putText(frame, f"LOGO:{logo_preset} {logo_scale:.0f}%", (10, 30), cv2.FONT_HERSHEY_SIMPLEX, 0.7, (0, 255, 255), 2, cv2.LINE_AA) - except Exception as _e: - self._log(f"⚠️ Logo preview error: {_e}") - - # ── CTA video overlay preview (every 5s) ── - if cta_enabled and cta_path and Path(cta_path).exists(): - try: - from app.core.logo_overlay import load_cta_frame_rgba, overlay_cta_on_frame, should_show_cta_at_time - if should_show_cta_at_time(float(sec), float(cta_interval), float(cta_duration)): - # CTA time loops within CTA video duration - _cta_time = float(sec) % float(cta_interval) - # If CTA video is shorter than interval, loop inside - _cta_rgba = load_cta_frame_rgba(cta_path, cta_time_sec=_cta_time, chromakey=cta_chromakey) - if _cta_rgba is not None: - frame = overlay_cta_on_frame( - frame_bgr=frame, - cta_rgba=_cta_rgba, - preset=cta_preset, - scale_pct=float(cta_scale), - opacity=float(cta_opacity), - custom_x_pct=float(cta_x_pct), - custom_y_pct=float(cta_y_pct), - margin_pct=2.0 - ) - cv2.putText(frame, f"CTA:{cta_preset} {cta_scale:.0f}% every {cta_interval:.0f}s", (10, 60), cv2.FONT_HERSHEY_SIMPLEX, 0.6, (0, 255, 0), 2, cv2.LINE_AA) - else: - cv2.putText(frame, f"CTA: hidden at {sec:.1f}s (interval {cta_interval:.0f}s)", (10, 60), cv2.FONT_HERSHEY_SIMPLEX, 0.5, (0, 165, 255), 1, cv2.LINE_AA) - except Exception as _e: - self._log(f"⚠️ CTA preview error: {_e}") - - if return_type == "base64": - _, buffer = cv2.imencode('.jpg', frame, [cv2.IMWRITE_JPEG_QUALITY, 85]) - return base64.b64encode(buffer).decode('utf-8') - - return cv2.cvtColor(frame, cv2.COLOR_BGR2RGB) - - def extract_subtitles_only( - self, - video_path: str, - mode: str = "asr", - source_lang: str = "zh", - region_preset: str = "custom", - custom_y_pct: float = 75.0, - custom_h_pct: float = 20.0, - custom_x_pct: float = 0.0, - custom_w_pct: float = 100.0 - ) -> Optional[Dict[str, Any]]: - """ - Runs Stage 0 -> Stage A -> Stage C. - Returns parsed subtitle blocks (original & translated) for Web Subtitle Editor. - """ - v_path = Path(video_path) - if not v_path.exists(): - self._log(f"❌ Video not found: {video_path}") - return None - - video_stem = v_path.stem - vtd = self.temp_dir / video_stem - vtd.mkdir(parents=True, exist_ok=True) - - self.progress_fn(5, "STAGE_0_PREPARE") - calc_region = self.calculate_region( - v_path, region_preset, custom_y_pct, custom_h_pct, custom_x_pct, custom_w_pct - ) - x, y, w, h = calc_region - - # 1. Extract audio - extracted_audio = vtd / "extracted_audio.wav" - self._log("⚡ Trích xuất âm thanh từ video gốc...") - cmd = [ - str(self.ffmpeg_path), "-y", "-i", str(v_path), - "-vn", "-acodec", "pcm_s16le", "-ar", "16000", "-ac", "1", - str(extracted_audio) - ] - subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True) - - # 2. OCR / ASR - self.progress_fn(25, "STAGE_A_OCR_ASR") - original_srt = vtd / "original.srt" - if mode == "ocr": - self._log("👁️ Quét chữ phụ đề bằng AI Vision Models...") - ocr_box = (x, y, w, h) if (w > 0 and h > 0) else None - ok = self.ocr_engine.scan_video_subtitles_to_srt(str(v_path), str(original_srt), blur_region=ocr_box) - if not ok or not original_srt.exists() or original_srt.stat().st_size < 10: - self._log("⚠️ OCR không phát hiện chữ -> Tự động chuyển sang Cloud ASR...") - ok = self.asr_engine.transcribe_audio_to_srt(str(extracted_audio), str(original_srt), source_lang=source_lang) - else: - self._log("🎙️ Nhận diện giọng nói bằng Cloud ASR (Groq Whisper Large-v3)...") - ok = self.asr_engine.transcribe_audio_to_srt(str(extracted_audio), str(original_srt), source_lang=source_lang) - - if not ok or not original_srt.exists() or original_srt.stat().st_size < 10: - raise RuntimeError("Không thể trích xuất phụ đề từ video.") - - # 3. Translation - self.progress_fn(50, "STAGE_C_TRANSLATION") - self._log("🌐 Dịch thuật bằng AI Translation Matrix...") - translated_srt = vtd / "translated.srt" - self._translate_srt_cloud(original_srt, translated_srt, source_lang=source_lang) - - orig_blocks = self._parse_srt_blocks(original_srt.read_text(encoding="utf-8", errors="ignore")) - trans_blocks = self._parse_srt_blocks(translated_srt.read_text(encoding="utf-8", errors="ignore")) - - combined = [] - for ob in orig_blocks: - tb = next((t for t in trans_blocks if t["id"] == ob["id"]), None) - combined.append({ - "id": ob["id"], - "timing": ob["timing"], - "start_ms": ob["start_ms"], - "end_ms": ob["end_ms"], - "original_text": ob["text"], - "vietnamese_text": tb["text"] if tb else ob["text"], - "sfx": "" - }) - - return { - "video_stem": video_stem, - "blocks": combined, - "original_srt": str(original_srt), - "translated_srt": str(translated_srt) - } - - def render_from_subtitles( - self, - video_path: str, - subtitles_data: List[Dict[str, Any]], - voice: str = "vi-VN-NamMinhNeural", - speed: float = 1.0, - pitch: int = 0, - volume: int = 100, - region_preset: str = "custom", - sub_mask_mode: str = "box", - sub_style: str = "motion_drip", - custom_y_pct: float = 75.0, - custom_h_pct: float = 20.0, - custom_x_pct: float = 0.0, - custom_w_pct: float = 100.0, - ducking_ratio: float = 0.18, - enable_sfx: bool = True, - mute_original_audio: bool = False, - # ── Logo overlay ── - logo_enabled: bool = False, - logo_path: str = "", - logo_preset: str = "bottom_right", - logo_scale: float = 15.0, - logo_opacity: float = 1.0, - logo_x_pct: float = 80.0, - logo_y_pct: float = 80.0, - logo_chromakey: bool = True, - # ── CTA video overlay (every 5s) ── - cta_enabled: bool = False, - cta_path: str = "", - cta_preset: str = "bottom_center", - cta_scale: float = 35.0, - cta_opacity: float = 1.0, - cta_x_pct: float = 50.0, - cta_y_pct: float = 85.0, - cta_interval: float = 5.0, - cta_duration: float = 2.0, - cta_chromakey: bool = True - ) -> Optional[str]: - """Completes TTS synthesis, Ducking audio mix, and video rendering from user-reviewed subtitles.""" - v_path = Path(video_path) - if not v_path.exists(): - return None - - video_stem = v_path.stem - vtd = self.temp_dir / video_stem - vtd.mkdir(parents=True, exist_ok=True) - - start_time = time.time() - calc_region = self.calculate_region( - v_path, region_preset, custom_y_pct, custom_h_pct, custom_x_pct, custom_w_pct - ) - x, y, w, h = calc_region - - # Write edited SRT - translated_srt = vtd / "translated.srt" - srt_lines = [] - for item in subtitles_data: - srt_lines.append(str(item["id"])) - srt_lines.append(item["timing"]) - vi_text = self.normalizer.normalize(item["vietnamese_text"]) if hasattr(self.normalizer, "normalize") else item["vietnamese_text"] - srt_lines.append(vi_text) - srt_lines.append("") - translated_srt.write_text("\n".join(srt_lines), encoding="utf-8") - - # 4. Cloud TTS - self.progress_fn(65, "STAGE_TTS_DUBBING") - self._log(f"🎙️ Tạo giọng đọc lồng tiếng ({voice}, speed={speed}x, volume={volume}%)...") - dubbing_wav = vtd / "dubbing.wav" - ok_tts = self.tts_engine.synthesize_srt_to_audio( - str(translated_srt), - str(dubbing_wav), - voice=voice, - speed=speed, - pitch=pitch, - volume=volume, - temp_dir=str(vtd / "tts_segments") - ) - if not ok_tts or not dubbing_wav.exists(): - raise RuntimeError("Lỗi tạo giọng đọc TTS.") - - # 5. SFX & Audio Ducking - self.progress_fn(80, "STAGE_D_AUDIO_MIX") - extracted_audio = vtd / "extracted_audio.wav" - if not extracted_audio.exists(): - cmd = [ - str(self.ffmpeg_path), "-y", "-i", str(v_path), - "-vn", "-acodec", "pcm_s16le", "-ar", "16000", "-ac", "1", - str(extracted_audio) - ] - subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True) - - mixed_audio = vtd / "mixed_final.wav" - if mute_original_audio: - self._log("🔇 Tắt hoàn toàn tiếng gốc — chỉ giữ giọng Việt (mute_original_audio=ON)") - # Dubbing wav đã được time-stretch canvas đúng duration, chỉ cần chuẩn hóa sample-rate - cmd_mix = [ - str(self.ffmpeg_path), "-y", - "-i", str(dubbing_wav), - "-ac", "2", "-ar", "48000", - str(mixed_audio) - ] - subprocess.run(cmd_mix, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True) - else: - self._log("🎚️ Trộn nhạc nền (Auto Ducking) + Giọng đọc thuyết minh...") - filter_complex = ( - f"[0:a]volume={ducking_ratio}[bg];" - f"[1:a]volume=1.0[dub];" - f"[bg][dub]amix=inputs=2:duration=longest:dropout_transition=2:normalize=0[aout]" - ) - cmd_mix = [ - str(self.ffmpeg_path), "-y", - "-i", str(extracted_audio), - "-i", str(dubbing_wav), - "-filter_complex", filter_complex, - "-map", "[aout]", - "-ac", "2", "-ar", "48000", - str(mixed_audio) - ] - subprocess.run(cmd_mix, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True) - - # 6. Render Video - self.progress_fn(90, "STAGE_D_RENDER") - self._log("🎬 Render video hoàn thiện: Xóa sub cũ & Đè sub tiếng Việt...") - final_output = self.output_dir / f"studio_final_{v_path.name}" - - # Probe dimensions with cv2 then ffprobe fallback (handles vertical & cv2-missing envs) - vw, vh = 1920, 1080 - try: - cap = cv2.VideoCapture(str(v_path)) - w_tmp = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH)) or 0 - h_tmp = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT)) or 0 - cap.release() - if w_tmp > 0 and h_tmp > 0: - vw, vh = w_tmp, h_tmp - else: - raise ValueError("cv2 returned 0") - except Exception: - try: - import shutil as _sh - _ffprobe = _sh.which("ffprobe") or str(Path(self.ffmpeg_path).parent / "ffprobe.exe") if Path(self.ffmpeg_path).exists() else "ffprobe" - if not Path(_ffprobe).exists(): - _ffprobe = "ffprobe" - _res = subprocess.run([_ffprobe, "-v", "error", "-select_streams", "v:0", "-show_entries", "stream=width,height", "-of", "json", str(v_path)], capture_output=True, text=True, timeout=5) - _j = json.loads(_res.stdout or "{}") - _ws = _j.get("streams", [{}])[0] - if _ws.get("width") and _ws.get("height"): - vw, vh = int(_ws["width"]), int(_ws["height"]) - except Exception: - pass - - translated_ass = vtd / "translated.ass" - self._convert_srt_to_ass(translated_srt, translated_ass, vw, vh, x, y, w, h, sub_style) - - ass_escaped = str(translated_ass).replace("\\", "/").replace(":", "\\:") - sub_filter = f"subtitles='{ass_escaped}'" - - vf_filters = [] - if w > 0 and h > 0: - if sub_mask_mode == "delogo": - safe_x = max(2, x) - safe_y = max(2, y) - safe_w = max(4, min(w, vw - safe_x - 2)) - safe_h = max(4, min(h, vh - safe_y - 2)) - vf_filters.append(f"delogo=x={safe_x}:y={safe_y}:w={safe_w}:h={safe_h}") - elif sub_mask_mode == "box": - vf_filters.append(f"drawbox=x={x}:y={y}:w={w}:h={h}:color=black@0.92:t=fill") - - # Determine logo & CTA overlay enabled (with fallback to bundled assets) - _logo_enabled = bool(logo_enabled and logo_path and Path(logo_path).exists()) - if logo_enabled and not _logo_enabled: - _fb = self.base_dir / "assets" / "logo_ins_drippy4.png" - if _fb.exists(): - _logo_enabled = True - logo_path = str(_fb) - _logo_path = Path(logo_path) if _logo_enabled else None - _cta_enabled = bool(cta_enabled and cta_path and Path(cta_path).exists()) - if cta_enabled and not _cta_enabled: - _fb2 = self.base_dir / "assets" / "cta_ins_drippy4.mp4" - if _fb2.exists(): - _cta_enabled = True - cta_path = str(_fb2) - _cta_path = Path(cta_path) if _cta_enabled else None - - # If any overlay enabled, we need filter_complex - if _logo_enabled or _cta_enabled: - if _logo_enabled: - self._log(f"🖼️ Overlay logo: {Path(logo_path).name} preset={logo_preset} scale={logo_scale}% opacity={logo_opacity} chromakey={logo_chromakey}") - if _cta_enabled: - self._log(f"🎬 Overlay CTA: {Path(cta_path).name} preset={cta_preset} scale={cta_scale}% every {cta_interval}s for {cta_duration}s chromakey={cta_chromakey}") - # Build filter parts step-by-step. Inputs: 0:video, 1:audio, 2:logo(if), 3:cta(if) - # Determine indices - _logo_idx = 2 if _logo_enabled else None - _cta_idx = None - if _cta_enabled: - _cta_idx = 3 if _logo_enabled else 2 - # Video preprocessing (delogo/box) -> [base] - _filter_parts = [] - if vf_filters and len([f for f in vf_filters if f != sub_filter]) > 0: - _pre_vf = ",".join([f for f in vf_filters if f != sub_filter]) - _filter_parts.append(f"[0:v]{_pre_vf}[base]") - _cur = "base" - else: - _filter_parts.append("[0:v]null[base]") - _cur = "base" - # Logo overlay - if _logo_enabled: - _logo_w_orig, _logo_h_orig = 2400, 1792 - try: - _t = cv2.imread(str(_logo_path), cv2.IMREAD_UNCHANGED) - if _t is not None: - _logo_h_orig, _logo_w_orig = _t.shape[:2] - except Exception: - pass - _target_w = vw * float(logo_scale) / 100.0 - _sf = max(0.02, min(0.5, _target_w / float(max(1, _logo_w_orig)))) - _logo_vf_parts = [] - if logo_chromakey: - _logo_vf_parts.append("colorkey=0x00FF00:0.3:0.1") - _logo_vf_parts.append("format=rgba") - _logo_vf_parts.append(f"scale=iw*{_sf:.4f}:ih*{_sf:.4f}:flags=lanczos") - if float(logo_opacity) < 0.99: - _logo_vf_parts.append(f"colorchannelmixer=aa={float(logo_opacity):.2f}") - _logo_vf = ",".join(_logo_vf_parts) - _m = 2.0 - if logo_preset == "top_left": - _lx, _ly = f"W*{_m/100:.3f}", f"H*{_m/100:.3f}" - elif logo_preset == "top_right": - _lx, _ly = f"W-w-W*{_m/100:.3f}", f"H*{_m/100:.3f}" - elif logo_preset == "bottom_left": - _lx, _ly = f"W*{_m/100:.3f}", f"H-h-H*{_m/100:.3f}" - elif logo_preset == "bottom_right": - _lx, _ly = f"W-w-W*{_m/100:.3f}", f"H-h-H*{_m/100:.3f}" - elif logo_preset == "center": - _lx, _ly = "(W-w)/2", "(H-h)/2" - elif logo_preset == "top_center": - _lx, _ly = "(W-w)/2", f"H*{_m/100:.3f}" - elif logo_preset == "bottom_center": - _lx, _ly = "(W-w)/2", f"H-h-H*{_m/100:.3f}" - elif logo_preset == "custom": - _lx, _ly = f"W*{float(logo_x_pct)/100:.4f}", f"H*{float(logo_y_pct)/100:.4f}" - else: - _lx, _ly = f"W-w-W*{_m/100:.3f}", f"H-h-H*{_m/100:.3f}" - _filter_parts.append(f"[{_logo_idx}:v]{_logo_vf}[logo]") - _next = "with_logo" if _cta_enabled or sub_filter else "v" - _filter_parts.append(f"[{_cur}][logo]overlay={_lx}:{_ly}:format=rgb[{_next}]") - _cur = _next - # CTA overlay (periodic every interval) - if _cta_enabled: - # Probe CTA size - _cta_w_orig, _cta_h_orig = 1080, 1920 - try: - _cap = cv2.VideoCapture(str(_cta_path)) - _cta_w_orig = int(_cap.get(cv2.CAP_PROP_FRAME_WIDTH)) or _cta_w_orig - _cta_h_orig = int(_cap.get(cv2.CAP_PROP_FRAME_HEIGHT)) or _cta_h_orig - _cap.release() - except Exception: - pass - # TikTok CTA bump: enforce minimum 42% width for mobile visibility - effective_cta_scale = max(float(cta_scale), 42.0) - _cta_target_w = vw * effective_cta_scale / 100.0 - _cta_sf = max(0.05, min(1.0, _cta_target_w / float(max(1, _cta_w_orig)))) - # Clamp height to 80% of video height to avoid overflow (CTA vertical on horizontal video) - _cta_target_h = _cta_h_orig * _cta_sf - _max_h = vh * 0.90 - if _cta_target_h > _max_h: - _cta_sf = _max_h / float(max(1, _cta_h_orig)) - _cta_target_w = _cta_w_orig * _cta_sf - _cta_vf_parts = [] - if cta_chromakey: - _cta_vf_parts.append("colorkey=0x00FF00:0.3:0.1") - _cta_vf_parts.append("format=rgba") - _cta_vf_parts.append(f"scale=iw*{_cta_sf:.4f}:ih*{_cta_sf:.4f}:flags=lanczos") - if float(cta_opacity) < 0.99: - _cta_vf_parts.append(f"colorchannelmixer=aa={float(cta_opacity):.2f}") - _cta_vf = ",".join(_cta_vf_parts) - _m2 = 2.0 - if cta_preset == "top_left": - _cx, _cy = f"W*{_m2/100:.3f}", f"H*{_m2/100:.3f}" - elif cta_preset == "top_right": - _cx, _cy = f"W-w-W*{_m2/100:.3f}", f"H*{_m2/100:.3f}" - elif cta_preset == "bottom_left": - _cx, _cy = f"W*{_m2/100:.3f}", f"H-h-H*{_m2/100:.3f}" - elif cta_preset == "bottom_right": - _cx, _cy = f"W-w-W*{_m2/100:.3f}", f"H-h-H*{_m2/100:.3f}" - elif cta_preset == "center": - _cx, _cy = "(W-w)/2", "(H-h)/2" - elif cta_preset == "top_center": - _cx, _cy = "(W-w)/2", f"H*{_m2/100:.3f}" - elif cta_preset == "bottom_center": - _cx, _cy = "(W-w)/2", f"H-h-H*{_m2/100:.3f}" - elif cta_preset == "custom": - _cx, _cy = f"W*{float(cta_x_pct)/100:.4f}", f"H*{float(cta_y_pct)/100:.4f}" - else: - _cx, _cy = "(W-w)/2", f"H-h-H*{_m2/100:.3f}" - _enable = f"lt(mod(t\\,{float(cta_interval)})\\,{float(cta_duration)})" - _filter_parts.append(f"[{_cta_idx}:v]{_cta_vf}[cta]") - _next2 = "v" if not sub_filter else "with_cta" - # Use escaped comma for FFmpeg enable expression - _filter_parts.append(f"[{_cur}][cta]overlay={_cx}:{_cy}:format=rgb:enable='{_enable}'[{_next2}]") - _cur = _next2 - # TikTok polish: CFR 30 + even dims + light sharpen BEFORE subtitles (keeps text razor sharp) - tiktok_polish = "fps=30:round=near,scale=trunc(iw/2)*2:trunc(ih/2)*2:flags=lanczos+accurate_rnd+full_chroma_int:sws_dither=ed,unsharp=3:3:0.35:3:3:0.0" - if sub_filter: - _filter_parts.append(f"[{_cur}]{tiktok_polish}[polished]") - _cur = "polished" - _filter_parts.append(f"[{_cur}]{sub_filter}[v]") - _cur = "v" - else: - _filter_parts.append(f"[{_cur}]{tiktok_polish}[v]") - _cur = "v" - _filter_complex = ";".join(_filter_parts) - # Build ffmpeg inputs: video, audio, logo(if), cta(if) - add -shortest when looped overlays present - cmd_inputs = [str(self.ffmpeg_path), "-y", "-fflags", "+genpts", "-avoid_negative_ts", "make_zero", "-i", str(v_path), "-i", str(mixed_audio)] - if _logo_enabled: - cmd_inputs += ["-loop", "1", "-i", str(_logo_path)] - if _cta_enabled: - cmd_inputs += ["-stream_loop", "999", "-i", str(_cta_path)] - _extra = ["-shortest"] if (_logo_enabled or _cta_enabled) else [] - # TikTok spec: High Profile yuv420p, 30 CFR, 48k AAC, faststart, accurate sync - tiktok_video_args = ["-r", "30", "-c:v", "libx264", "-preset", "medium", "-crf", "18", "-profile:v", "high", "-level", "4.1", "-pix_fmt", "yuv420p", "-g", "60", "-keyint_min", "30", "-sc_threshold", "0", "-x264-params", "ref=4:bframes=2:me=hex:subme=7:psy=1:psy-rd=0.8:aq-mode=2", "-colorspace", "bt709", "-color_primaries", "bt709", "-color_trc", "bt709", "-color_range", "tv"] - tiktok_audio_args = ["-c:a", "aac", "-profile:a", "aac_low", "-ar", "48000", "-ac", "2", "-b:a", "192k", "-af", "aresample=async=1:min_hard_comp=0.100000:first_pts=0"] - tiktok_mux_args = ["-movflags", "+faststart", "-fflags", "+genpts", "-max_interleave_delta", "100M", "-vsync", "cfr", "-fps_mode", "cfr"] - cmd_render = cmd_inputs + ["-filter_complex", _filter_complex, "-map", "[v]", "-map", "1:a:0"] + _extra + tiktok_video_args + tiktok_audio_args + tiktok_mux_args + ["-shortest", str(final_output)] - res = subprocess.run(cmd_render, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True) - if res.returncode != 0: - self._log(f"⚠️ Overlay render failed ({res.stderr[:300]}), fallback to TikTok-spec normal render...") - # Rebuild VF with TikTok polish inserted before subs (same as non-overlay branch) - tiktok_polish_fb = "fps=30:round=near,scale=trunc(iw/2)*2:trunc(ih/2)*2:flags=lanczos+accurate_rnd+full_chroma_int:sws_dither=ed,unsharp=3:3:0.35:3:3:0.0" - vf_without_sub_fb = [f for f in vf_filters if f != sub_filter] - ordered_fb = [] - if vf_without_sub_fb: - ordered_fb.extend(vf_without_sub_fb) - ordered_fb.append(tiktok_polish_fb) - ordered_fb.append(sub_filter) - final_vf = ",".join(ordered_fb) - cmd_render = [ - str(self.ffmpeg_path), "-y", - "-fflags", "+genpts", "-avoid_negative_ts", "make_zero", - "-i", str(v_path), - "-i", str(mixed_audio), - "-vf", final_vf, - "-map", "0:v:0", "-map", "1:a:0", - "-r", "30", - "-c:v", "libx264", "-preset", "medium", "-crf", "18", - "-profile:v", "high", "-level", "4.1", "-pix_fmt", "yuv420p", - "-g", "60", "-keyint_min", "30", "-sc_threshold", "0", - "-x264-params", "ref=4:bframes=2:me=hex:subme=7:psy=1:psy-rd=0.8:aq-mode=2", - "-colorspace", "bt709", "-color_primaries", "bt709", "-color_trc", "bt709", "-color_range", "tv", - "-c:a", "aac", "-profile:a", "aac_low", "-ar", "48000", "-ac", "2", "-b:a", "192k", - "-af", "aresample=async=1:min_hard_comp=0.100000:first_pts=0", - "-movflags", "+faststart", - "-fflags", "+genpts", "-max_interleave_delta", "100M", - "-vsync", "cfr", "-fps_mode", "cfr", - "-shortest", - str(final_output) - ] - res = subprocess.run(cmd_render, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True) - if res.returncode != 0: - self._log(f"⚠️ Fallback render không mask: {res.stderr[:120]}") - fallback_cmd = [ - str(self.ffmpeg_path), "-y", - "-fflags", "+genpts", - "-i", str(v_path), - "-i", str(mixed_audio), - "-vf", f"{tiktok_polish_fb},{sub_filter}", - "-map", "0:v:0", "-map", "1:a:0", - "-r", "30", - "-c:v", "libx264", "-preset", "medium", "-crf", "18", - "-profile:v", "high", "-pix_fmt", "yuv420p", - "-c:a", "aac", "-profile:a", "aac_low", "-ar", "48000", "-ac", "2", "-b:a", "192k", - "-movflags", "+faststart", - "-shortest", - str(final_output) - ] - subprocess.run(fallback_cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True) - else: - # TikTok polish inserted before subtitles for sharpness + CFR + even dims - tiktok_polish = "fps=30:round=near,scale=trunc(iw/2)*2:trunc(ih/2)*2:flags=lanczos+accurate_rnd+full_chroma_int:sws_dither=ed,unsharp=3:3:0.35:3:3:0.0" - # Order: mask (delogo/box) -> polish -> subtitles - vf_without_sub = [f for f in vf_filters if f != sub_filter] - # Rebuild vf in TikTok-optimal order - ordered_vf = [] - if vf_without_sub: - ordered_vf.extend(vf_without_sub) - ordered_vf.append(tiktok_polish) - ordered_vf.append(sub_filter) - final_vf = ",".join(ordered_vf) - - cmd_render = [ - str(self.ffmpeg_path), "-y", - "-fflags", "+genpts", "-avoid_negative_ts", "make_zero", - "-i", str(v_path), - "-i", str(mixed_audio), - "-vf", final_vf, - "-map", "0:v:0", "-map", "1:a:0", - "-r", "30", - "-c:v", "libx264", "-preset", "medium", "-crf", "18", - "-profile:v", "high", "-level", "4.1", "-pix_fmt", "yuv420p", - "-g", "60", "-keyint_min", "30", "-sc_threshold", "0", - "-x264-params", "ref=4:bframes=2:me=hex:subme=7:psy=1:psy-rd=0.8:aq-mode=2", - "-colorspace", "bt709", "-color_primaries", "bt709", "-color_trc", "bt709", "-color_range", "tv", - "-c:a", "aac", "-profile:a", "aac_low", "-ar", "48000", "-ac", "2", "-b:a", "192k", - "-af", "aresample=async=1:min_hard_comp=0.100000:first_pts=0", - "-movflags", "+faststart", - "-fflags", "+genpts", "-max_interleave_delta", "100M", - "-vsync", "cfr", "-fps_mode", "cfr", - "-shortest", - str(final_output) - ] - res = subprocess.run(cmd_render, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True) - if res.returncode != 0: - self._log(f"⚠️ Fallback render không mask: {res.stderr[:120]} | retry with minimal TikTok spec") - # Minimal fallback: at least ensure TikTok audio/video spec - fallback_cmd = [ - str(self.ffmpeg_path), "-y", - "-fflags", "+genpts", - "-i", str(v_path), - "-i", str(mixed_audio), - "-vf", f"{tiktok_polish},{sub_filter}", - "-map", "0:v:0", "-map", "1:a:0", - "-r", "30", - "-c:v", "libx264", "-preset", "medium", "-crf", "18", - "-profile:v", "high", "-pix_fmt", "yuv420p", - "-c:a", "aac", "-profile:a", "aac_low", "-ar", "48000", "-ac", "2", "-b:a", "192k", - "-movflags", "+faststart", - "-shortest", - str(final_output) - ] - subprocess.run(fallback_cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True) - - total_elapsed = time.time() - start_time - self.progress_fn(100, "DONE") - self._log(f"🎉 HOÀN THÀNH XUẤT SẮC TRONG {total_elapsed:.1f}s!") - self._log(f"📁 Video đã lưu tại: {final_output}") - return str(final_output) - - def run_video( - self, - video_path: str, - mode: str = "asr", - source_lang: str = "zh", - voice: str = "vi-VN-NamMinhNeural", - speed: float = 1.0, - pitch: int = 0, - volume: int = 100, - region_preset: str = "custom", - sub_mask_mode: str = "box", - sub_style: str = "motion_drip", - custom_y_pct: float = 75.0, - custom_h_pct: float = 20.0, - custom_x_pct: float = 0.0, - custom_w_pct: float = 100.0, - ducking_ratio: float = 0.18, - enable_sfx: bool = True, - mute_original_audio: bool = False, - logo_enabled: bool = False, - logo_path: str = "", - logo_preset: str = "bottom_right", - logo_scale: float = 15.0, - logo_opacity: float = 1.0, - logo_x_pct: float = 80.0, - logo_y_pct: float = 80.0, - logo_chromakey: bool = True, - cta_enabled: bool = False, - cta_path: str = "", - cta_preset: str = "bottom_center", - cta_scale: float = 35.0, - cta_opacity: float = 1.0, - cta_x_pct: float = 50.0, - cta_y_pct: float = 85.0, - cta_interval: float = 5.0, - cta_duration: float = 2.0, - cta_chromakey: bool = True - ) -> Optional[str]: - """Full 1-Click End-to-End Automatic Dubbing Pipeline.""" - v_path = Path(video_path) - if not v_path.exists(): - self._log(f"❌ Video not found: {video_path}") - return None - - # Register into SQLite Job DB - try: - self.job_manager.register_job(str(v_path)) - except Exception: - pass - - try: - # 1. Extract subtitles - subs_res = self.extract_subtitles_only( - video_path=video_path, - mode=mode, - source_lang=source_lang, - region_preset=region_preset, - custom_y_pct=custom_y_pct, - custom_h_pct=custom_h_pct, - custom_x_pct=custom_x_pct, - custom_w_pct=custom_w_pct - ) - if not subs_res: - return None - - # 2. Render from subtitles - return self.render_from_subtitles( - video_path=video_path, - subtitles_data=subs_res["blocks"], - voice=voice, - speed=speed, - pitch=pitch, - volume=volume, - region_preset=region_preset, - sub_mask_mode=sub_mask_mode, - sub_style=sub_style, - custom_y_pct=custom_y_pct, - custom_h_pct=custom_h_pct, - custom_x_pct=custom_x_pct, - custom_w_pct=custom_w_pct, - ducking_ratio=ducking_ratio, - enable_sfx=enable_sfx, - mute_original_audio=mute_original_audio, - logo_enabled=logo_enabled, - logo_path=logo_path, - logo_preset=logo_preset, - logo_scale=logo_scale, - logo_opacity=logo_opacity, - logo_x_pct=logo_x_pct, - logo_y_pct=logo_y_pct, - logo_chromakey=logo_chromakey, - cta_enabled=cta_enabled, - cta_path=cta_path, - cta_preset=cta_preset, - cta_scale=cta_scale, - cta_opacity=cta_opacity, - cta_x_pct=cta_x_pct, - cta_y_pct=cta_y_pct, - cta_interval=cta_interval, - cta_duration=cta_duration, - cta_chromakey=cta_chromakey - ) - except Exception as e: - self._log(f"❌ LỖI PIPELINE: {str(e)}") - self.progress_fn(0, "FAILED") - return None - - def _convert_srt_to_ass( - self, - srt_path: Path, - ass_path: Path, - vw: int, - vh: int, - box_x: int, - box_y: int, - box_w: int, - box_h: int, - sub_style: str = "motion_drip" - ): - """Converts SRT to styled ASS format for TikTok/Shorts typography.""" - content = srt_path.read_text(encoding="utf-8", errors="ignore") - - # TikTok safe zone: keep subtitle inside 82% height, 7% bottom margin, larger font for mobile legibility - safe_margin_v = max(42, int(vh * 0.07)) - raw_margin_v = int(vh - (box_y + box_h * 0.75)) if box_h > 0 else safe_margin_v - margin_v = max(safe_margin_v, raw_margin_v) - # Clamp to avoid bottom UI overlap - max_bottom = int(vh * 0.12) - if margin_v < max_bottom: - margin_v = max_bottom - # Larger font for TikTok: vertical 56-72, horizontal 48-64 - if box_h > 0: - base = int(box_h * 0.52) - if vh > vw: # vertical TikTok - font_size = max(42, min(72, base)) - else: - font_size = max(38, min(64, base)) - else: - font_size = 58 if vh > vw else 50 - - # Typography Color & Outline palettes - TikTok optimized for high compression - if sub_style == "motion_drip" or sub_style == "yellow_bold": - font_color = "&H0000FFFF" # Bright Yellow - outline_color = "&H00000000" # Pure Black - back_color = "&H99000000" - outline = 5.0 - shadow = 2.2 - elif sub_style == "neon_cyan": - font_color = "&H00FFFF00" # Cyan - outline_color = "&H00000000" - back_color = "&H99000000" - outline = 5.0 - shadow = 2.2 - elif sub_style == "capsule_tag": - font_color = "&H00FFFFFF" # White - outline_color = "&H00111111" - back_color = "&HBB000000" - outline = 3.2 - shadow = 0.0 - else: # white_bold - font_color = "&H00FFFFFF" # Crisp White - outline_color = "&H00000000" - back_color = "&H99000000" - outline = 4.8 - shadow = 2.0 - - ass_header = f"""[Script Info] -ScriptType: v4.00+ -PlayResX: {vw} -PlayResY: {vh} -ScaledBorderAndShadow: yes - -[V4+ Styles] -Format: Name, Fontname, Fontsize, PrimaryColour, SecondaryColour, OutlineColour, BackColour, Bold, Italic, Underline, StrikeOut, ScaleX, ScaleY, Spacing, Angle, BorderStyle, Outline, Shadow, Alignment, MarginL, MarginR, MarginV, Encoding -Style: Default,DejaVu Sans,{font_size},{font_color},&H000000FF,{outline_color},{back_color},1,0,0,0,100,100,0,0,1,{outline},{shadow},2,40,40,{margin_v},1 - -[Events] -Format: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text -""" - - events = [] - pattern = r"(\d+)\s+(\d{2}:\d{2}:\d{2}[.,]\d{3})\s*-->\s*(\d{2}:\d{2}:\d{2}[.,]\d{3})\s*\n(.*?)(?=\n\s*\d+\s+\d{2}:\d{2}:\d{2}[.,]\d{3}\s*-->|\Z)" - for m in re.finditer(pattern, content, re.DOTALL): - start_str = m.group(2).strip().replace(",", ".") - end_str = m.group(3).strip().replace(",", ".") - - s_parts = start_str.split(":") - e_parts = end_str.split(":") - s_ass = f"{int(s_parts[0])}:{s_parts[1]}:{float(s_parts[2]):05.2f}" - e_ass = f"{int(e_parts[0])}:{e_parts[1]}:{float(e_parts[2]):05.2f}" - - text = " ".join(line.strip() for line in m.group(4).splitlines() if line.strip()) - if text: - events.append(f"Dialogue: 0,{s_ass},{e_ass},Default,,0,0,0,,{text}") - - ass_path.write_text(ass_header + "\n".join(events), encoding="utf-8") - - def _parse_srt_blocks(self, content: str) -> List[Dict]: - pattern = r"(\d+)\s+(\d{2}:\d{2}:\d{2}[.,]\d{3})\s*-->\s*(\d{2}:\d{2}:\d{2}[.,]\d{3})\s*\n(.*?)(?=\n\s*\d+\s+\d{2}:\d{2}:\d{2}[.,]\d{3}\s*-->|\Z)" - blocks = [] - for m in re.finditer(pattern, content, re.DOTALL): - text = " ".join(line.strip() for line in m.group(4).splitlines() if line.strip()) - start_str = m.group(2).strip() - end_str = m.group(3).strip() - if text: - blocks.append({ - "id": int(m.group(1)), - "timing": f"{start_str} --> {end_str}", - "text": text, - "start_ms": self._ts_to_ms(start_str), - "end_ms": self._ts_to_ms(end_str), - }) - return blocks - - @staticmethod - def _ts_to_ms(ts: str) -> int: - ts = ts.strip().replace(".", ",") - m = re.match(r"(\d+):(\d+):(\d+)[,](\d+)", ts) - if m: - h, mins, s, ms = map(int, m.groups()) - return ((h * 3600 + mins * 60 + s) * 1000) + ms - return 0 - - def _build_system_prompt(self) -> str: - glossary_sample = ", ".join([f"{k} -> {v}" for k, v in list(self.glossary.items())[:35]]) - return ( - "Bạn là chuyên gia dịch thuật video Sneaker, Thời trang Streetwear, Review sản phẩm từ tiếng Trung/Anh sang tiếng Việt tự nhiên, sành điệu, bắt trend Gen Z.\n" - "QUY TẮC BẮT BUỘC:\n" - "1. Dịch từng dòng theo cấu trúc: [N] Câu dịch tiếng Việt hoàn chỉnh.\n" - "2. Giữ nguyên thuật ngữ & thương hiệu tiếng Anh (Nike, Jordan, Yeezy, BAPE, Supreme, Rick Owens, outfit, fit, drip, full box, collab, signature...). \n" - f"3. Bắt buộc áp dụng từ điển chuyên ngành: {glossary_sample}\n" - "4. Dịch ĐẦY ĐỦ Ý NGHĨA trọn vẹn của câu, giữ đủ các từ khoá, KHÔNG bỏ lửng hay cắt cụt mất từ ở cuối câu.\n" - "5. KHÔNG giải thích, CHỈ trả về danh sách các dòng [N] Tiếng Việt." - ) - - def _translate_srt_cloud(self, srt_in: Path, srt_out: Path, source_lang: str = "zh"): - content = srt_in.read_text(encoding="utf-8", errors="ignore") - blocks = self._parse_srt_blocks(content) - - if not blocks: - srt_out.write_text(content, encoding="utf-8") - return - - system_prompt = self._build_system_prompt() - validator = TranslationValidator() - trans_map = {} - - # Phase 1: Batch translation in chunks of 20 - chunk_size = 20 - for i in range(0, len(blocks), chunk_size): - chunk = blocks[i:i + chunk_size] - chunk_result = self._translate_chunk_with_retry(chunk, system_prompt, validator, max_retries=2) - trans_map.update(chunk_result) - - # Phase 2: Retry missing / Chinese-leaked blocks - missing_blocks = [b for b in blocks if (b["id"] not in trans_map or has_chinese(trans_map.get(b["id"], "")))] - if missing_blocks: - self._log(f"🔄 Đang hoàn thiện nốt {len(missing_blocks)} câu dịch còn lại...") - retry_result = self._translate_chunk_with_retry(missing_blocks, system_prompt, validator, max_retries=2) - trans_map.update(retry_result) - - # Phase 3: Google Translate fallback - still_missing = [b for b in blocks if (b["id"] not in trans_map or has_chinese(trans_map.get(b["id"], "")))] - if still_missing: - self._log(f"🌐 {len(still_missing)} câu cần fallback Google Translate...") - google_fb = GoogleTranslateFallback(log_fn=self._log) - google_result = google_fb.translate_blocks(still_missing) - for bid_str, text in google_result.items(): - trans_map[int(bid_str)] = text - - # Phase 4: Post-editing - self._log("✨ Chạy Post-Editor sửa lỗi dịch...") - for b in blocks: - bid = b["id"] - if bid in trans_map: - trans_map[bid] = post_edit_translation(b["text"], trans_map[bid]) - - # Phase 5: Quality Guard - self._log("🛡️ Quality Guard kiểm tra chất lượng bản dịch...") - guard = TranslationQualityGuard(min_score=70) - guard_dict = {str(b["id"]): trans_map.get(b["id"], b["text"]) for b in blocks} - fixed_dict, quality_report = guard.audit_and_fix(blocks, guard_dict) - - for b in blocks: - bid_str = str(b["id"]) - if bid_str in fixed_dict: - trans_map[b["id"]] = fixed_dict[bid_str] - - # Phase 6: Timing compaction - self._log("⏱️ Rút gọn câu dịch cho vừa timeline TTS...") - compact_dict = {str(b["id"]): trans_map.get(b["id"], b["text"]) for b in blocks} - compact_dict = compact_blocks_for_tts(blocks, compact_dict) - for b in blocks: - bid_str = str(b["id"]) - if bid_str in compact_dict: - trans_map[b["id"]] = compact_dict[bid_str] - - # Phase 7: Normalize + Output SRT - out_lines = [] - for b in blocks: - vi_text = trans_map.get(b["id"], b["text"]) - vi_text = self.normalizer.normalize(vi_text) if hasattr(self.normalizer, "normalize") else vi_text - out_lines.append(str(b["id"])) - out_lines.append(b["timing"]) - out_lines.append(vi_text) - out_lines.append("") - - srt_out.write_text("\n".join(out_lines), encoding="utf-8") - self._log(f"✅ Dịch hoàn tất {len(blocks)} câu với 7 bước kiểm tra chất lượng.") - - def _translate_chunk_with_retry( - self, - chunk: List[Dict], - system_prompt: str, - validator: TranslationValidator, - max_retries: int = 2 - ) -> Dict[int, str]: - prompt_lines = [f"[{b['id']}] {b['text']}" for b in chunk] - full_transcript = "\n".join(prompt_lines) - result = {} - - for attempt in range(max_retries + 1): - raw_res = self._direct_25_model_translate(system_prompt, full_transcript) - cleaned_res = re.sub(r".*?", "", raw_res, flags=re.DOTALL).strip() - - attempt_map = {} - for line in cleaned_res.splitlines(): - m = re.match(r"^\s*\[(\d+)\]\s*(.*)$", line.strip()) - if m: - attempt_map[int(m.group(1))] = m.group(2).strip() - - val_dict = {str(b["id"]): attempt_map.get(b["id"], "") for b in chunk if b["id"] in attempt_map} - val_chunk = [b for b in chunk if b["id"] in attempt_map] - - if val_chunk and val_dict: - is_valid, error_msg = validator.validate(val_chunk, val_dict) - if is_valid: - result.update(attempt_map) - return result - else: - for bid, text in attempt_map.items(): - src_block = next((b for b in chunk if b["id"] == bid), None) - if src_block: - ok, _ = validator.validate_single_block(bid, src_block["text"], text) - if ok: - result[bid] = text - else: - result.update(attempt_map) - - return result - - def _direct_25_model_translate(self, system_prompt: str, user_content: str) -> str: - messages = [ - {"role": "system", "content": system_prompt}, - {"role": "user", "content": user_content} - ] - - # 1. Groq - groq_keys = self.asr_engine.groq_keys - for model in ["qwen/qwen3.6-27b", "openai/gpt-oss-120b", "openai/gpt-oss-20b", "groq/compound"]: - for key in groq_keys: - try: - res = requests.post( - "https://api.groq.com/openai/v1/chat/completions", - headers={"Authorization": f"Bearer {key}", "Content-Type": "application/json"}, - json={"model": model, "messages": messages, "temperature": 0.3, "max_tokens": 4096}, - timeout=30 - ) - if res.status_code == 200: - content = res.json()["choices"][0]["message"]["content"].strip() - content = re.sub(r".*?", "", content, flags=re.DOTALL).strip() - if len(content) > 15: - return content - except Exception: - pass - - # 2. Google Gemini - gemini_keys = self.ocr_engine.gemini_keys - for model in ["gemini-3.6-flash", "gemini-flash-latest", "gemini-3.5-flash", "gemini-pro-latest"]: - for key in gemini_keys: - try: - url = f"https://generativelanguage.googleapis.com/v1beta/models/{model}:generateContent?key={key}" - payload = { - "contents": [{"parts": [{"text": f"{system_prompt}\n\n{user_content}"}]}], - "generationConfig": {"temperature": 0.3, "maxOutputTokens": 4096} - } - res = requests.post(url, json=payload, timeout=30) - if res.status_code == 200: - content = res.json()["candidates"][0]["content"]["parts"][0]["text"].strip() - if len(content) > 15: - return content - except Exception: - pass - - # 3. OpenRouter - or_keys = self.ocr_engine.openrouter_keys - for model in ["nvidia/nemotron-3-super-120b-a12b:free", "nvidia/nemotron-3.5-lightning:free", "openai/gpt-oss-20b:free"]: - for key in or_keys: - try: - res = requests.post( - "https://openrouter.ai/api/v1/chat/completions", - headers={"Authorization": f"Bearer {key}", "Content-Type": "application/json", "HTTP-Referer": "https://trungsangviet.local", "X-Title": "TrungSangViet"}, - json={"model": model, "messages": messages, "temperature": 0.3, "max_tokens": 4096}, - timeout=35 - ) - if res.status_code == 200: - content = res.json()["choices"][0]["message"]["content"].strip() - if len(content) > 15: - return content - except Exception: - pass - - return user_content +""" +app/core/cloud_pipeline.py +────────────────────────── +Pure Cloud-Native Headless Pipeline Coordinator. +Features: +1. High-Definition Video Preview Frame Generator using FFmpeg (100% Reliable). +2. Multi-Model ASR Matrix (Groq Whisper Large V3 / OpenRouter / Gemini). +3. AI Vision Models OCR with automatic ASR fallback. +4. Active 2026 AI Translation Models (Groq Qwen 3.6 / GPT-OSS 120B / Gemini Flash / Nemotron 3). +5. Glossary & Slang Mapping (Fashion, Sneaker, Streetwear, Chinese Slang). +6. 7-Step Quality Guard & Semantic Validation & Timing Compaction. +7. Subtitle Editing Hooks (extract_subtitles_only & render_from_subtitles). +8. ASS Subtitle Engine with TikTok / Shorts Typography & Word Jump Styles. +9. 100% Solid/Delogo Masking Box to eliminate old Chinese subtitles. +10. Studio SFX & Audio Ducking Mixer. +11. SQLite JobManager Integration for state persistence. +""" + +import os +import re +import sys +import json +import time +import base64 +import shutil +import cv2 +import requests +import subprocess +from pathlib import Path +from typing import Optional, Callable, Dict, Any, Tuple, List + +from app.core.cloud_asr import CloudASREngine +from app.core.cloud_ocr import CloudOCREngine +from app.core.cloud_tts import CloudTTSEngine +from app.core.vietnamese_text_normalizer import VietnameseTextNormalizer +from app.core.translation_validator import TranslationValidator +from app.core.translation_quality_guard import TranslationQualityGuard +from app.core.translation_post_editor import post_edit_translation +from app.core.google_translate_fallback import GoogleTranslateFallback +from app.core.subtitle_compactor import compact_blocks_for_tts +from app.core.job_manager import JobManager +from app.core.studio_sfx import generate_sfx_clip + + +def has_chinese(text: str) -> bool: + """Returns True if string contains CJK Chinese characters.""" + return bool(re.search(r"[\u4e00-\u9fff]", text)) + + +class CloudPipeline: + def __init__( + self, + base_dir: Optional[Path] = None, + log_callback: Optional[Callable[[str], None]] = None, + progress_callback: Optional[Callable[[int, str], None]] = None, + ffmpeg_path: str = "ffmpeg" + ): + self.base_dir = Path(base_dir or Path(__file__).resolve().parents[2]) + self.output_dir = self.base_dir / "output" + self.temp_dir = self.base_dir / "temp" + self.output_dir.mkdir(parents=True, exist_ok=True) + self.temp_dir.mkdir(parents=True, exist_ok=True) + + self.log_fn = log_callback or print + self.progress_fn = progress_callback or (lambda pct, stage: None) + self.ffmpeg_path = ffmpeg_path + + self.asr_engine = CloudASREngine(log_fn=self._log) + self.ocr_engine = CloudOCREngine(log_fn=self._log) + self.tts_engine = CloudTTSEngine(log_fn=self._log, ffmpeg_path=self.ffmpeg_path) + self.normalizer = VietnameseTextNormalizer() + self.job_manager = JobManager.instance() + self.glossary = self._load_glossary() + + def _log(self, msg: str): + self.log_fn(f"[Cloud Pipeline] {msg}") + + def _load_glossary(self) -> Dict[str, str]: + glossary_path = self.base_dir / "glossary.json" + if glossary_path.exists(): + try: + data = json.loads(glossary_path.read_text(encoding="utf-8")) + return data.get("glossary", {}) + except Exception as e: + self._log(f"⚠️ Lỗi đọc glossary.json: {e}") + return {} + + def calculate_region( + self, + video_path: Path, + region_preset: str = "custom", + custom_y_pct: float = 75.0, + custom_h_pct: float = 20.0, + custom_x_pct: float = 0.0, + custom_w_pct: float = 100.0 + ) -> Tuple[int, int, int, int]: + vw, vh = 1920, 1080 + try: + cap = cv2.VideoCapture(str(video_path)) + w = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH)) + h = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT)) + cap.release() + if w > 0 and h > 0: + vw, vh = w, h + except Exception: + pass + + if region_preset == "bottom_25": + x = 0 + w = vw + y = int(vh * 0.72) + h = int(vh * 0.24) + elif region_preset == "bottom_15": + x = 0 + w = vw + y = int(vh * 0.82) + h = int(vh * 0.15) + elif region_preset == "middle_25": + x = 0 + w = vw + y = int(vh * 0.38) + h = int(vh * 0.24) + elif region_preset == "none": + return (0, 0, 0, 0) + elif region_preset == "custom": + x = int(vw * (custom_x_pct / 100.0)) + w = int(vw * (custom_w_pct / 100.0)) + y = int(vh * (custom_y_pct / 100.0)) + h = int(vh * (custom_h_pct / 100.0)) + else: + return (0, 0, 0, 0) + + x = max(0, min(vw - 2, x)) + y = max(0, min(vh - 2, y)) + w = max(2, min(vw - x, w)) + h = max(2, min(vh - y, h)) + return (x, y, w, h) + + def generate_preview_frame( + self, + video_path: str, + region_preset: str = "custom", + custom_y_pct: float = 75.0, + custom_h_pct: float = 20.0, + custom_x_pct: float = 0.0, + custom_w_pct: float = 100.0, + sec: float = 2.0, + return_type: str = "rgb", + # ── Logo overlay params ── + logo_enabled: bool = False, + logo_path: str = "", + logo_preset: str = "bottom_right", + logo_scale: float = 15.0, + logo_opacity: float = 1.0, + logo_x_pct: float = 80.0, + logo_y_pct: float = 80.0, + logo_chromakey: bool = True, + # ── CTA video overlay params (appears every 5s) ── + cta_enabled: bool = False, + cta_path: str = "", + cta_preset: str = "bottom_center", + cta_scale: float = 35.0, + cta_opacity: float = 1.0, + cta_x_pct: float = 50.0, + cta_y_pct: float = 85.0, + cta_interval: float = 5.0, + cta_duration: float = 2.0, + cta_chromakey: bool = True + ): + """Extracts a frame at timestamp sec and overlays bounding box + logo + CTA preview.""" + v_file = Path(video_path) + if not v_file.exists(): + return None + + raw_frame_path = self.temp_dir / f"raw_frame_{int(time.time()*1000)}.jpg" + cmd = [ + str(self.ffmpeg_path), "-ss", str(sec), "-y", + "-i", str(v_file), + "-frames:v", "1", "-q:v", "2", + str(raw_frame_path) + ] + try: + subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True) + frame = cv2.imread(str(raw_frame_path)) + if raw_frame_path.exists(): + raw_frame_path.unlink() + except Exception: + frame = None + + if frame is None: + cap = cv2.VideoCapture(str(v_file)) + fps = cap.get(cv2.CAP_PROP_FPS) or 25.0 + target_frame = max(0, int(sec * fps)) + cap.set(cv2.CAP_PROP_POS_FRAMES, target_frame) + ret, frame = cap.read() + cap.release() + + if frame is None: + return None + + x, y, w, h = self.calculate_region( + v_file, region_preset, custom_y_pct, custom_h_pct, custom_x_pct, custom_w_pct + ) + + if w > 0 and h > 0: + overlay = frame.copy() + cv2.rectangle(overlay, (x, y), (x + w, y + h), (0, 0, 255), -1) + cv2.addWeighted(overlay, 0.35, frame, 0.65, 0, frame) + cv2.rectangle(frame, (x, y), (x + w, y + h), (0, 255, 255), 3) + + label = "VUNG CHE SUB CU & QUET OCR" + font = cv2.FONT_HERSHEY_SIMPLEX + font_scale = max(0.5, frame.shape[1] / 1200.0) + thickness = 2 + (tw, th), _ = cv2.getTextSize(label, font, font_scale, thickness) + + label_y = max(th + 10, y - 8) + cv2.rectangle(frame, (x, label_y - th - 6), (x + tw + 10, label_y + 4), (0, 0, 0), -1) + cv2.putText(frame, label, (x + 5, label_y), font, font_scale, (0, 255, 255), thickness, cv2.LINE_AA) + + # ── Logo overlay preview ── + if logo_enabled and logo_path and Path(logo_path).exists(): + try: + from app.core.logo_overlay import load_logo_rgba, overlay_logo_on_frame + _logo_rgba = load_logo_rgba(logo_path, chromakey=logo_chromakey) + if _logo_rgba is not None: + frame = overlay_logo_on_frame( + frame_bgr=frame, + logo_rgba=_logo_rgba, + preset=logo_preset, + scale_pct=float(logo_scale), + opacity=float(logo_opacity), + custom_x_pct=float(logo_x_pct), + custom_y_pct=float(logo_y_pct), + margin_pct=2.0 + ) + cv2.putText(frame, f"LOGO:{logo_preset} {logo_scale:.0f}%", (10, 30), cv2.FONT_HERSHEY_SIMPLEX, 0.7, (0, 255, 255), 2, cv2.LINE_AA) + except Exception as _e: + self._log(f"⚠️ Logo preview error: {_e}") + + # ── CTA video overlay preview (every 5s) ── + if cta_enabled and cta_path and Path(cta_path).exists(): + try: + from app.core.logo_overlay import load_cta_frame_rgba, overlay_cta_on_frame, should_show_cta_at_time + if should_show_cta_at_time(float(sec), float(cta_interval), float(cta_duration)): + # CTA time loops within CTA video duration + _cta_time = float(sec) % float(cta_interval) + # If CTA video is shorter than interval, loop inside + _cta_rgba = load_cta_frame_rgba(cta_path, cta_time_sec=_cta_time, chromakey=cta_chromakey) + if _cta_rgba is not None: + frame = overlay_cta_on_frame( + frame_bgr=frame, + cta_rgba=_cta_rgba, + preset=cta_preset, + scale_pct=float(cta_scale), + opacity=float(cta_opacity), + custom_x_pct=float(cta_x_pct), + custom_y_pct=float(cta_y_pct), + margin_pct=2.0 + ) + cv2.putText(frame, f"CTA:{cta_preset} {cta_scale:.0f}% every {cta_interval:.0f}s", (10, 60), cv2.FONT_HERSHEY_SIMPLEX, 0.6, (0, 255, 0), 2, cv2.LINE_AA) + else: + cv2.putText(frame, f"CTA: hidden at {sec:.1f}s (interval {cta_interval:.0f}s)", (10, 60), cv2.FONT_HERSHEY_SIMPLEX, 0.5, (0, 165, 255), 1, cv2.LINE_AA) + except Exception as _e: + self._log(f"⚠️ CTA preview error: {_e}") + + if return_type == "base64": + _, buffer = cv2.imencode('.jpg', frame, [cv2.IMWRITE_JPEG_QUALITY, 85]) + return base64.b64encode(buffer).decode('utf-8') + + return cv2.cvtColor(frame, cv2.COLOR_BGR2RGB) + + def extract_subtitles_only( + self, + video_path: str, + mode: str = "asr", + source_lang: str = "zh", + region_preset: str = "custom", + custom_y_pct: float = 75.0, + custom_h_pct: float = 20.0, + custom_x_pct: float = 0.0, + custom_w_pct: float = 100.0 + ) -> Optional[Dict[str, Any]]: + """ + Runs Stage 0 -> Stage A -> Stage C. + Returns parsed subtitle blocks (original & translated) for Web Subtitle Editor. + """ + v_path = Path(video_path) + if not v_path.exists(): + self._log(f"❌ Video not found: {video_path}") + return None + + video_stem = v_path.stem + vtd = self.temp_dir / video_stem + vtd.mkdir(parents=True, exist_ok=True) + + self.progress_fn(5, "STAGE_0_PREPARE") + calc_region = self.calculate_region( + v_path, region_preset, custom_y_pct, custom_h_pct, custom_x_pct, custom_w_pct + ) + x, y, w, h = calc_region + + # 1. Extract audio + extracted_audio = vtd / "extracted_audio.wav" + self._log("⚡ Trích xuất âm thanh từ video gốc...") + cmd = [ + str(self.ffmpeg_path), "-y", "-i", str(v_path), + "-vn", "-acodec", "pcm_s16le", "-ar", "16000", "-ac", "1", + str(extracted_audio) + ] + subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True) + + # 2. OCR / ASR + self.progress_fn(25, "STAGE_A_OCR_ASR") + original_srt = vtd / "original.srt" + if mode == "ocr": + self._log("👁️ Quét chữ phụ đề bằng AI Vision Models...") + ocr_box = (x, y, w, h) if (w > 0 and h > 0) else None + ok = self.ocr_engine.scan_video_subtitles_to_srt(str(v_path), str(original_srt), blur_region=ocr_box) + if not ok or not original_srt.exists() or original_srt.stat().st_size < 10: + self._log("⚠️ OCR không phát hiện chữ -> Tự động chuyển sang Cloud ASR...") + ok = self.asr_engine.transcribe_audio_to_srt(str(extracted_audio), str(original_srt), source_lang=source_lang) + else: + self._log("🎙️ Nhận diện giọng nói bằng Cloud ASR (Groq Whisper Large-v3)...") + ok = self.asr_engine.transcribe_audio_to_srt(str(extracted_audio), str(original_srt), source_lang=source_lang) + + if not ok or not original_srt.exists() or original_srt.stat().st_size < 10: + raise RuntimeError("Không thể trích xuất phụ đề từ video.") + + # 3. Translation + self.progress_fn(50, "STAGE_C_TRANSLATION") + self._log("🌐 Dịch thuật bằng AI Translation Matrix...") + translated_srt = vtd / "translated.srt" + self._translate_srt_cloud(original_srt, translated_srt, source_lang=source_lang) + + orig_blocks = self._parse_srt_blocks(original_srt.read_text(encoding="utf-8", errors="ignore")) + trans_blocks = self._parse_srt_blocks(translated_srt.read_text(encoding="utf-8", errors="ignore")) + + combined = [] + for ob in orig_blocks: + tb = next((t for t in trans_blocks if t["id"] == ob["id"]), None) + combined.append({ + "id": ob["id"], + "timing": ob["timing"], + "start_ms": ob["start_ms"], + "end_ms": ob["end_ms"], + "original_text": ob["text"], + "vietnamese_text": tb["text"] if tb else ob["text"], + "sfx": "" + }) + + return { + "video_stem": video_stem, + "blocks": combined, + "original_srt": str(original_srt), + "translated_srt": str(translated_srt) + } + + def render_from_subtitles( + self, + video_path: str, + subtitles_data: List[Dict[str, Any]], + voice: str = "vi-VN-NamMinhNeural", + speed: float = 1.0, + pitch: int = 0, + volume: int = 100, + region_preset: str = "custom", + sub_mask_mode: str = "box", + sub_style: str = "motion_drip", + custom_y_pct: float = 75.0, + custom_h_pct: float = 20.0, + custom_x_pct: float = 0.0, + custom_w_pct: float = 100.0, + ducking_ratio: float = 0.18, + enable_sfx: bool = True, + mute_original_audio: bool = False, + # ── Logo overlay ── + logo_enabled: bool = False, + logo_path: str = "", + logo_preset: str = "bottom_right", + logo_scale: float = 15.0, + logo_opacity: float = 1.0, + logo_x_pct: float = 80.0, + logo_y_pct: float = 80.0, + logo_chromakey: bool = True, + # ── CTA video overlay (every 5s) ── + cta_enabled: bool = False, + cta_path: str = "", + cta_preset: str = "bottom_center", + cta_scale: float = 35.0, + cta_opacity: float = 1.0, + cta_x_pct: float = 50.0, + cta_y_pct: float = 85.0, + cta_interval: float = 5.0, + cta_duration: float = 2.0, + cta_chromakey: bool = True + ) -> Optional[str]: + """Completes TTS synthesis, Ducking audio mix, and video rendering from user-reviewed subtitles.""" + v_path = Path(video_path) + if not v_path.exists(): + return None + + video_stem = v_path.stem + vtd = self.temp_dir / video_stem + vtd.mkdir(parents=True, exist_ok=True) + + start_time = time.time() + calc_region = self.calculate_region( + v_path, region_preset, custom_y_pct, custom_h_pct, custom_x_pct, custom_w_pct + ) + x, y, w, h = calc_region + + # Write edited SRT + translated_srt = vtd / "translated.srt" + srt_lines = [] + for item in subtitles_data: + srt_lines.append(str(item["id"])) + srt_lines.append(item["timing"]) + vi_text = self.normalizer.normalize(item["vietnamese_text"]) if hasattr(self.normalizer, "normalize") else item["vietnamese_text"] + srt_lines.append(vi_text) + srt_lines.append("") + translated_srt.write_text("\n".join(srt_lines), encoding="utf-8") + + # 4. Cloud TTS + self.progress_fn(65, "STAGE_TTS_DUBBING") + self._log(f"🎙️ Tạo giọng đọc lồng tiếng ({voice}, speed={speed}x, volume={volume}%)...") + dubbing_wav = vtd / "dubbing.wav" + ok_tts = self.tts_engine.synthesize_srt_to_audio( + str(translated_srt), + str(dubbing_wav), + voice=voice, + speed=speed, + pitch=pitch, + volume=volume, + temp_dir=str(vtd / "tts_segments") + ) + if not ok_tts or not dubbing_wav.exists(): + raise RuntimeError("Lỗi tạo giọng đọc TTS.") + + # 5. SFX & Audio Ducking + self.progress_fn(80, "STAGE_D_AUDIO_MIX") + extracted_audio = vtd / "extracted_audio.wav" + if not extracted_audio.exists(): + cmd = [ + str(self.ffmpeg_path), "-y", "-i", str(v_path), + "-vn", "-acodec", "pcm_s16le", "-ar", "16000", "-ac", "1", + str(extracted_audio) + ] + subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True) + + mixed_audio = vtd / "mixed_final.wav" + if mute_original_audio: + self._log("🔇 Tắt hoàn toàn tiếng gốc — chỉ giữ giọng Việt (mute_original_audio=ON)") + # Dubbing wav đã được time-stretch canvas đúng duration, chỉ cần chuẩn hóa sample-rate + cmd_mix = [ + str(self.ffmpeg_path), "-y", + "-i", str(dubbing_wav), + "-ac", "2", "-ar", "48000", + str(mixed_audio) + ] + subprocess.run(cmd_mix, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True) + else: + self._log("🎚️ Trộn nhạc nền (Auto Ducking) + Giọng đọc thuyết minh...") + filter_complex = ( + f"[0:a]volume={ducking_ratio}[bg];" + f"[1:a]volume=1.0[dub];" + f"[bg][dub]amix=inputs=2:duration=longest:dropout_transition=2:normalize=0[aout]" + ) + cmd_mix = [ + str(self.ffmpeg_path), "-y", + "-i", str(extracted_audio), + "-i", str(dubbing_wav), + "-filter_complex", filter_complex, + "-map", "[aout]", + "-ac", "2", "-ar", "48000", + str(mixed_audio) + ] + subprocess.run(cmd_mix, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True) + + # 6. Render Video + self.progress_fn(90, "STAGE_D_RENDER") + self._log("🎬 Render video hoàn thiện: Xóa sub cũ & Đè sub tiếng Việt...") + final_output = self.output_dir / f"studio_final_{v_path.name}" + + # Probe dimensions with cv2 then ffprobe fallback (handles vertical & cv2-missing envs) + vw, vh = 1920, 1080 + try: + cap = cv2.VideoCapture(str(v_path)) + w_tmp = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH)) or 0 + h_tmp = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT)) or 0 + cap.release() + if w_tmp > 0 and h_tmp > 0: + vw, vh = w_tmp, h_tmp + else: + raise ValueError("cv2 returned 0") + except Exception: + try: + import shutil as _sh + _ffprobe = _sh.which("ffprobe") or str(Path(self.ffmpeg_path).parent / "ffprobe.exe") if Path(self.ffmpeg_path).exists() else "ffprobe" + if not Path(_ffprobe).exists(): + _ffprobe = "ffprobe" + _res = subprocess.run([_ffprobe, "-v", "error", "-select_streams", "v:0", "-show_entries", "stream=width,height", "-of", "json", str(v_path)], capture_output=True, text=True, timeout=5) + _j = json.loads(_res.stdout or "{}") + _ws = _j.get("streams", [{}])[0] + if _ws.get("width") and _ws.get("height"): + vw, vh = int(_ws["width"]), int(_ws["height"]) + except Exception: + pass + + translated_ass = vtd / "translated.ass" + self._convert_srt_to_ass(translated_srt, translated_ass, vw, vh, x, y, w, h, sub_style) + + ass_escaped = str(translated_ass).replace("\\", "/").replace(":", "\\:") + sub_filter = f"subtitles='{ass_escaped}'" + + vf_filters = [] + if w > 0 and h > 0: + if sub_mask_mode == "delogo": + safe_x = max(2, x) + safe_y = max(2, y) + safe_w = max(4, min(w, vw - safe_x - 2)) + safe_h = max(4, min(h, vh - safe_y - 2)) + vf_filters.append(f"delogo=x={safe_x}:y={safe_y}:w={safe_w}:h={safe_h}") + elif sub_mask_mode == "box": + vf_filters.append(f"drawbox=x={x}:y={y}:w={w}:h={h}:color=black@0.92:t=fill") + + # Determine logo & CTA overlay enabled (with fallback to bundled assets) + _logo_enabled = bool(logo_enabled and logo_path and Path(logo_path).exists()) + if logo_enabled and not _logo_enabled: + _fb = self.base_dir / "assets" / "logo_ins_drippy4.png" + if _fb.exists(): + _logo_enabled = True + logo_path = str(_fb) + _logo_path = Path(logo_path) if _logo_enabled else None + _cta_enabled = bool(cta_enabled and cta_path and Path(cta_path).exists()) + if cta_enabled and not _cta_enabled: + _fb2 = self.base_dir / "assets" / "cta_ins_drippy4.mp4" + if _fb2.exists(): + _cta_enabled = True + cta_path = str(_fb2) + _cta_path = Path(cta_path) if _cta_enabled else None + + # If any overlay enabled, we need filter_complex + if _logo_enabled or _cta_enabled: + if _logo_enabled: + self._log(f"🖼️ Overlay logo: {Path(logo_path).name} preset={logo_preset} scale={logo_scale}% opacity={logo_opacity} chromakey={logo_chromakey}") + if _cta_enabled: + self._log(f"🎬 Overlay CTA: {Path(cta_path).name} preset={cta_preset} scale={cta_scale}% every {cta_interval}s for {cta_duration}s chromakey={cta_chromakey}") + # Build filter parts step-by-step. Inputs: 0:video, 1:audio, 2:logo(if), 3:cta(if) + # Determine indices + _logo_idx = 2 if _logo_enabled else None + _cta_idx = None + if _cta_enabled: + _cta_idx = 3 if _logo_enabled else 2 + # Video preprocessing (delogo/box) -> [base] + _filter_parts = [] + if vf_filters and len([f for f in vf_filters if f != sub_filter]) > 0: + _pre_vf = ",".join([f for f in vf_filters if f != sub_filter]) + _filter_parts.append(f"[0:v]{_pre_vf}[base]") + _cur = "base" + else: + _filter_parts.append("[0:v]null[base]") + _cur = "base" + # Logo overlay + if _logo_enabled: + _logo_w_orig, _logo_h_orig = 2400, 1792 + try: + _t = cv2.imread(str(_logo_path), cv2.IMREAD_UNCHANGED) + if _t is not None: + _logo_h_orig, _logo_w_orig = _t.shape[:2] + except Exception: + pass + _target_w = vw * float(logo_scale) / 100.0 + _sf = max(0.02, min(0.5, _target_w / float(max(1, _logo_w_orig)))) + _logo_vf_parts = [] + if logo_chromakey: + _logo_vf_parts.append("colorkey=0x00FF00:0.3:0.1") + _logo_vf_parts.append("format=rgba") + _logo_vf_parts.append(f"scale=iw*{_sf:.4f}:ih*{_sf:.4f}:flags=lanczos") + if float(logo_opacity) < 0.99: + _logo_vf_parts.append(f"colorchannelmixer=aa={float(logo_opacity):.2f}") + _logo_vf = ",".join(_logo_vf_parts) + _m = 2.0 + if logo_preset == "top_left": + _lx, _ly = f"W*{_m/100:.3f}", f"H*{_m/100:.3f}" + elif logo_preset == "top_right": + _lx, _ly = f"W-w-W*{_m/100:.3f}", f"H*{_m/100:.3f}" + elif logo_preset == "bottom_left": + _lx, _ly = f"W*{_m/100:.3f}", f"H-h-H*{_m/100:.3f}" + elif logo_preset == "bottom_right": + _lx, _ly = f"W-w-W*{_m/100:.3f}", f"H-h-H*{_m/100:.3f}" + elif logo_preset == "center": + _lx, _ly = "(W-w)/2", "(H-h)/2" + elif logo_preset == "top_center": + _lx, _ly = "(W-w)/2", f"H*{_m/100:.3f}" + elif logo_preset == "bottom_center": + _lx, _ly = "(W-w)/2", f"H-h-H*{_m/100:.3f}" + elif logo_preset == "custom": + _lx, _ly = f"W*{float(logo_x_pct)/100:.4f}", f"H*{float(logo_y_pct)/100:.4f}" + else: + _lx, _ly = f"W-w-W*{_m/100:.3f}", f"H-h-H*{_m/100:.3f}" + _filter_parts.append(f"[{_logo_idx}:v]{_logo_vf}[logo]") + _next = "with_logo" if _cta_enabled or sub_filter else "v" + _filter_parts.append(f"[{_cur}][logo]overlay={_lx}:{_ly}:format=rgb[{_next}]") + _cur = _next + # CTA overlay (periodic every interval) + if _cta_enabled: + # Probe CTA size + _cta_w_orig, _cta_h_orig = 1080, 1920 + try: + _cap = cv2.VideoCapture(str(_cta_path)) + _cta_w_orig = int(_cap.get(cv2.CAP_PROP_FRAME_WIDTH)) or _cta_w_orig + _cta_h_orig = int(_cap.get(cv2.CAP_PROP_FRAME_HEIGHT)) or _cta_h_orig + _cap.release() + except Exception: + pass + # TikTok CTA bump: enforce minimum 42% width for mobile visibility + effective_cta_scale = max(float(cta_scale), 42.0) + _cta_target_w = vw * effective_cta_scale / 100.0 + _cta_sf = max(0.05, min(1.0, _cta_target_w / float(max(1, _cta_w_orig)))) + # Clamp height to 80% of video height to avoid overflow (CTA vertical on horizontal video) + _cta_target_h = _cta_h_orig * _cta_sf + _max_h = vh * 0.90 + if _cta_target_h > _max_h: + _cta_sf = _max_h / float(max(1, _cta_h_orig)) + _cta_target_w = _cta_w_orig * _cta_sf + _cta_vf_parts = [] + if cta_chromakey: + _cta_vf_parts.append("colorkey=0x00FF00:0.3:0.1") + _cta_vf_parts.append("format=rgba") + _cta_vf_parts.append(f"scale=iw*{_cta_sf:.4f}:ih*{_cta_sf:.4f}:flags=lanczos") + if float(cta_opacity) < 0.99: + _cta_vf_parts.append(f"colorchannelmixer=aa={float(cta_opacity):.2f}") + _cta_vf = ",".join(_cta_vf_parts) + _m2 = 2.0 + if cta_preset == "top_left": + _cx, _cy = f"W*{_m2/100:.3f}", f"H*{_m2/100:.3f}" + elif cta_preset == "top_right": + _cx, _cy = f"W-w-W*{_m2/100:.3f}", f"H*{_m2/100:.3f}" + elif cta_preset == "bottom_left": + _cx, _cy = f"W*{_m2/100:.3f}", f"H-h-H*{_m2/100:.3f}" + elif cta_preset == "bottom_right": + _cx, _cy = f"W-w-W*{_m2/100:.3f}", f"H-h-H*{_m2/100:.3f}" + elif cta_preset == "center": + _cx, _cy = "(W-w)/2", "(H-h)/2" + elif cta_preset == "top_center": + _cx, _cy = "(W-w)/2", f"H*{_m2/100:.3f}" + elif cta_preset == "bottom_center": + _cx, _cy = "(W-w)/2", f"H-h-H*{_m2/100:.3f}" + elif cta_preset == "custom": + _cx, _cy = f"W*{float(cta_x_pct)/100:.4f}", f"H*{float(cta_y_pct)/100:.4f}" + else: + _cx, _cy = "(W-w)/2", f"H-h-H*{_m2/100:.3f}" + _enable = f"lt(mod(t\\,{float(cta_interval)})\\,{float(cta_duration)})" + _filter_parts.append(f"[{_cta_idx}:v]{_cta_vf}[cta]") + _next2 = "v" if not sub_filter else "with_cta" + # Use escaped comma for FFmpeg enable expression + _filter_parts.append(f"[{_cur}][cta]overlay={_cx}:{_cy}:format=rgb:enable='{_enable}'[{_next2}]") + _cur = _next2 + # TikTok polish: CFR 30 + even dims + light sharpen BEFORE subtitles (keeps text razor sharp) + tiktok_polish = "fps=30:round=near,scale=trunc(iw/2)*2:trunc(ih/2)*2:flags=lanczos+accurate_rnd+full_chroma_int:sws_dither=ed,unsharp=3:3:0.35:3:3:0.0" + if sub_filter: + _filter_parts.append(f"[{_cur}]{tiktok_polish}[polished]") + _cur = "polished" + _filter_parts.append(f"[{_cur}]{sub_filter}[v]") + _cur = "v" + else: + _filter_parts.append(f"[{_cur}]{tiktok_polish}[v]") + _cur = "v" + _filter_complex = ";".join(_filter_parts) + # Build ffmpeg inputs: video, audio, logo(if), cta(if) - add -shortest when looped overlays present + cmd_inputs = [str(self.ffmpeg_path), "-y", "-fflags", "+genpts", "-avoid_negative_ts", "make_zero", "-i", str(v_path), "-i", str(mixed_audio)] + if _logo_enabled: + cmd_inputs += ["-loop", "1", "-i", str(_logo_path)] + if _cta_enabled: + cmd_inputs += ["-stream_loop", "999", "-i", str(_cta_path)] + _extra = ["-shortest"] if (_logo_enabled or _cta_enabled) else [] + # TikTok spec: High Profile yuv420p, 30 CFR, 48k AAC, faststart, accurate sync + tiktok_video_args = ["-r", "30", "-c:v", "libx264", "-preset", "medium", "-crf", "18", "-profile:v", "high", "-level", "4.1", "-pix_fmt", "yuv420p", "-g", "60", "-keyint_min", "30", "-sc_threshold", "0", "-x264-params", "ref=4:bframes=2:me=hex:subme=7:psy=1:psy-rd=0.8:aq-mode=2", "-colorspace", "bt709", "-color_primaries", "bt709", "-color_trc", "bt709", "-color_range", "tv"] + tiktok_audio_args = ["-c:a", "aac", "-profile:a", "aac_low", "-ar", "48000", "-ac", "2", "-b:a", "192k", "-af", "aresample=async=1:min_hard_comp=0.100000:first_pts=0"] + tiktok_mux_args = ["-movflags", "+faststart", "-fflags", "+genpts", "-max_interleave_delta", "100M", "-vsync", "cfr", "-fps_mode", "cfr"] + cmd_render = cmd_inputs + ["-filter_complex", _filter_complex, "-map", "[v]", "-map", "1:a:0"] + _extra + tiktok_video_args + tiktok_audio_args + tiktok_mux_args + ["-shortest", str(final_output)] + res = subprocess.run(cmd_render, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True) + if res.returncode != 0: + self._log(f"⚠️ Overlay render failed ({res.stderr[:300]}), fallback to TikTok-spec normal render...") + # Rebuild VF with TikTok polish inserted before subs (same as non-overlay branch) + tiktok_polish_fb = "fps=30:round=near,scale=trunc(iw/2)*2:trunc(ih/2)*2:flags=lanczos+accurate_rnd+full_chroma_int:sws_dither=ed,unsharp=3:3:0.35:3:3:0.0" + vf_without_sub_fb = [f for f in vf_filters if f != sub_filter] + ordered_fb = [] + if vf_without_sub_fb: + ordered_fb.extend(vf_without_sub_fb) + ordered_fb.append(tiktok_polish_fb) + ordered_fb.append(sub_filter) + final_vf = ",".join(ordered_fb) + cmd_render = [ + str(self.ffmpeg_path), "-y", + "-fflags", "+genpts", "-avoid_negative_ts", "make_zero", + "-i", str(v_path), + "-i", str(mixed_audio), + "-vf", final_vf, + "-map", "0:v:0", "-map", "1:a:0", + "-r", "30", + "-c:v", "libx264", "-preset", "medium", "-crf", "18", + "-profile:v", "high", "-level", "4.1", "-pix_fmt", "yuv420p", + "-g", "60", "-keyint_min", "30", "-sc_threshold", "0", + "-x264-params", "ref=4:bframes=2:me=hex:subme=7:psy=1:psy-rd=0.8:aq-mode=2", + "-colorspace", "bt709", "-color_primaries", "bt709", "-color_trc", "bt709", "-color_range", "tv", + "-c:a", "aac", "-profile:a", "aac_low", "-ar", "48000", "-ac", "2", "-b:a", "192k", + "-af", "aresample=async=1:min_hard_comp=0.100000:first_pts=0", + "-movflags", "+faststart", + "-fflags", "+genpts", "-max_interleave_delta", "100M", + "-vsync", "cfr", "-fps_mode", "cfr", + "-shortest", + str(final_output) + ] + res = subprocess.run(cmd_render, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True) + if res.returncode != 0: + self._log(f"⚠️ Fallback render không mask: {res.stderr[:120]}") + fallback_cmd = [ + str(self.ffmpeg_path), "-y", + "-fflags", "+genpts", + "-i", str(v_path), + "-i", str(mixed_audio), + "-vf", f"{tiktok_polish_fb},{sub_filter}", + "-map", "0:v:0", "-map", "1:a:0", + "-r", "30", + "-c:v", "libx264", "-preset", "medium", "-crf", "18", + "-profile:v", "high", "-pix_fmt", "yuv420p", + "-c:a", "aac", "-profile:a", "aac_low", "-ar", "48000", "-ac", "2", "-b:a", "192k", + "-movflags", "+faststart", + "-shortest", + str(final_output) + ] + subprocess.run(fallback_cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True) + else: + # TikTok polish inserted before subtitles for sharpness + CFR + even dims + tiktok_polish = "fps=30:round=near,scale=trunc(iw/2)*2:trunc(ih/2)*2:flags=lanczos+accurate_rnd+full_chroma_int:sws_dither=ed,unsharp=3:3:0.35:3:3:0.0" + # Order: mask (delogo/box) -> polish -> subtitles + vf_without_sub = [f for f in vf_filters if f != sub_filter] + # Rebuild vf in TikTok-optimal order + ordered_vf = [] + if vf_without_sub: + ordered_vf.extend(vf_without_sub) + ordered_vf.append(tiktok_polish) + ordered_vf.append(sub_filter) + final_vf = ",".join(ordered_vf) + + cmd_render = [ + str(self.ffmpeg_path), "-y", + "-fflags", "+genpts", "-avoid_negative_ts", "make_zero", + "-i", str(v_path), + "-i", str(mixed_audio), + "-vf", final_vf, + "-map", "0:v:0", "-map", "1:a:0", + "-r", "30", + "-c:v", "libx264", "-preset", "medium", "-crf", "18", + "-profile:v", "high", "-level", "4.1", "-pix_fmt", "yuv420p", + "-g", "60", "-keyint_min", "30", "-sc_threshold", "0", + "-x264-params", "ref=4:bframes=2:me=hex:subme=7:psy=1:psy-rd=0.8:aq-mode=2", + "-colorspace", "bt709", "-color_primaries", "bt709", "-color_trc", "bt709", "-color_range", "tv", + "-c:a", "aac", "-profile:a", "aac_low", "-ar", "48000", "-ac", "2", "-b:a", "192k", + "-af", "aresample=async=1:min_hard_comp=0.100000:first_pts=0", + "-movflags", "+faststart", + "-fflags", "+genpts", "-max_interleave_delta", "100M", + "-vsync", "cfr", "-fps_mode", "cfr", + "-shortest", + str(final_output) + ] + res = subprocess.run(cmd_render, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True) + if res.returncode != 0: + self._log(f"⚠️ Fallback render không mask: {res.stderr[:120]} | retry with minimal TikTok spec") + # Minimal fallback: at least ensure TikTok audio/video spec + fallback_cmd = [ + str(self.ffmpeg_path), "-y", + "-fflags", "+genpts", + "-i", str(v_path), + "-i", str(mixed_audio), + "-vf", f"{tiktok_polish},{sub_filter}", + "-map", "0:v:0", "-map", "1:a:0", + "-r", "30", + "-c:v", "libx264", "-preset", "medium", "-crf", "18", + "-profile:v", "high", "-pix_fmt", "yuv420p", + "-c:a", "aac", "-profile:a", "aac_low", "-ar", "48000", "-ac", "2", "-b:a", "192k", + "-movflags", "+faststart", + "-shortest", + str(final_output) + ] + subprocess.run(fallback_cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True) + + total_elapsed = time.time() - start_time + self.progress_fn(100, "DONE") + self._log(f"🎉 HOÀN THÀNH XUẤT SẮC TRONG {total_elapsed:.1f}s!") + self._log(f"📁 Video đã lưu tại: {final_output}") + return str(final_output) + + def run_video( + self, + video_path: str, + mode: str = "asr", + source_lang: str = "zh", + voice: str = "vi-VN-NamMinhNeural", + speed: float = 1.0, + pitch: int = 0, + volume: int = 100, + region_preset: str = "custom", + sub_mask_mode: str = "box", + sub_style: str = "motion_drip", + custom_y_pct: float = 75.0, + custom_h_pct: float = 20.0, + custom_x_pct: float = 0.0, + custom_w_pct: float = 100.0, + ducking_ratio: float = 0.18, + enable_sfx: bool = True, + mute_original_audio: bool = False, + logo_enabled: bool = False, + logo_path: str = "", + logo_preset: str = "bottom_right", + logo_scale: float = 15.0, + logo_opacity: float = 1.0, + logo_x_pct: float = 80.0, + logo_y_pct: float = 80.0, + logo_chromakey: bool = True, + cta_enabled: bool = False, + cta_path: str = "", + cta_preset: str = "bottom_center", + cta_scale: float = 35.0, + cta_opacity: float = 1.0, + cta_x_pct: float = 50.0, + cta_y_pct: float = 85.0, + cta_interval: float = 5.0, + cta_duration: float = 2.0, + cta_chromakey: bool = True + ) -> Optional[str]: + """Full 1-Click End-to-End Automatic Dubbing Pipeline.""" + v_path = Path(video_path) + if not v_path.exists(): + self._log(f"❌ Video not found: {video_path}") + return None + + # Register into SQLite Job DB + try: + self.job_manager.register_job(str(v_path)) + except Exception: + pass + + try: + # 1. Extract subtitles + subs_res = self.extract_subtitles_only( + video_path=video_path, + mode=mode, + source_lang=source_lang, + region_preset=region_preset, + custom_y_pct=custom_y_pct, + custom_h_pct=custom_h_pct, + custom_x_pct=custom_x_pct, + custom_w_pct=custom_w_pct + ) + if not subs_res: + return None + + # 2. Render from subtitles + return self.render_from_subtitles( + video_path=video_path, + subtitles_data=subs_res["blocks"], + voice=voice, + speed=speed, + pitch=pitch, + volume=volume, + region_preset=region_preset, + sub_mask_mode=sub_mask_mode, + sub_style=sub_style, + custom_y_pct=custom_y_pct, + custom_h_pct=custom_h_pct, + custom_x_pct=custom_x_pct, + custom_w_pct=custom_w_pct, + ducking_ratio=ducking_ratio, + enable_sfx=enable_sfx, + mute_original_audio=mute_original_audio, + logo_enabled=logo_enabled, + logo_path=logo_path, + logo_preset=logo_preset, + logo_scale=logo_scale, + logo_opacity=logo_opacity, + logo_x_pct=logo_x_pct, + logo_y_pct=logo_y_pct, + logo_chromakey=logo_chromakey, + cta_enabled=cta_enabled, + cta_path=cta_path, + cta_preset=cta_preset, + cta_scale=cta_scale, + cta_opacity=cta_opacity, + cta_x_pct=cta_x_pct, + cta_y_pct=cta_y_pct, + cta_interval=cta_interval, + cta_duration=cta_duration, + cta_chromakey=cta_chromakey + ) + except Exception as e: + self._log(f"❌ LỖI PIPELINE: {str(e)}") + self.progress_fn(0, "FAILED") + return None + + def _convert_srt_to_ass( + self, + srt_path: Path, + ass_path: Path, + vw: int, + vh: int, + box_x: int, + box_y: int, + box_w: int, + box_h: int, + sub_style: str = "motion_drip" + ): + """Converts SRT to styled ASS format for TikTok/Shorts typography.""" + content = srt_path.read_text(encoding="utf-8", errors="ignore") + + # TikTok safe zone: keep subtitle inside 82% height, 7% bottom margin, larger font for mobile legibility + safe_margin_v = max(42, int(vh * 0.07)) + raw_margin_v = int(vh - (box_y + box_h * 0.75)) if box_h > 0 else safe_margin_v + margin_v = max(safe_margin_v, raw_margin_v) + # Clamp to avoid bottom UI overlap + max_bottom = int(vh * 0.12) + if margin_v < max_bottom: + margin_v = max_bottom + # Larger font for TikTok: vertical 56-72, horizontal 48-64 + if box_h > 0: + base = int(box_h * 0.52) + if vh > vw: # vertical TikTok + font_size = max(42, min(72, base)) + else: + font_size = max(38, min(64, base)) + else: + font_size = 58 if vh > vw else 50 + + # Typography Color & Outline palettes - TikTok optimized for high compression + if sub_style == "motion_drip" or sub_style == "yellow_bold": + font_color = "&H0000FFFF" # Bright Yellow + outline_color = "&H00000000" # Pure Black + back_color = "&H99000000" + outline = 5.0 + shadow = 2.2 + elif sub_style == "neon_cyan": + font_color = "&H00FFFF00" # Cyan + outline_color = "&H00000000" + back_color = "&H99000000" + outline = 5.0 + shadow = 2.2 + elif sub_style == "capsule_tag": + font_color = "&H00FFFFFF" # White + outline_color = "&H00111111" + back_color = "&HBB000000" + outline = 3.2 + shadow = 0.0 + else: # white_bold + font_color = "&H00FFFFFF" # Crisp White + outline_color = "&H00000000" + back_color = "&H99000000" + outline = 4.8 + shadow = 2.0 + + ass_header = f"""[Script Info] +ScriptType: v4.00+ +PlayResX: {vw} +PlayResY: {vh} +ScaledBorderAndShadow: yes + +[V4+ Styles] +Format: Name, Fontname, Fontsize, PrimaryColour, SecondaryColour, OutlineColour, BackColour, Bold, Italic, Underline, StrikeOut, ScaleX, ScaleY, Spacing, Angle, BorderStyle, Outline, Shadow, Alignment, MarginL, MarginR, MarginV, Encoding +Style: Default,DejaVu Sans,{font_size},{font_color},&H000000FF,{outline_color},{back_color},1,0,0,0,100,100,0,0,1,{outline},{shadow},2,40,40,{margin_v},1 + +[Events] +Format: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text +""" + + events = [] + pattern = r"(\d+)\s+(\d{2}:\d{2}:\d{2}[.,]\d{3})\s*-->\s*(\d{2}:\d{2}:\d{2}[.,]\d{3})\s*\n(.*?)(?=\n\s*\d+\s+\d{2}:\d{2}:\d{2}[.,]\d{3}\s*-->|\Z)" + for m in re.finditer(pattern, content, re.DOTALL): + start_str = m.group(2).strip().replace(",", ".") + end_str = m.group(3).strip().replace(",", ".") + + s_parts = start_str.split(":") + e_parts = end_str.split(":") + s_ass = f"{int(s_parts[0])}:{s_parts[1]}:{float(s_parts[2]):05.2f}" + e_ass = f"{int(e_parts[0])}:{e_parts[1]}:{float(e_parts[2]):05.2f}" + + text = " ".join(line.strip() for line in m.group(4).splitlines() if line.strip()) + if text: + events.append(f"Dialogue: 0,{s_ass},{e_ass},Default,,0,0,0,,{text}") + + ass_path.write_text(ass_header + "\n".join(events), encoding="utf-8") + + def _parse_srt_blocks(self, content: str) -> List[Dict]: + pattern = r"(\d+)\s+(\d{2}:\d{2}:\d{2}[.,]\d{3})\s*-->\s*(\d{2}:\d{2}:\d{2}[.,]\d{3})\s*\n(.*?)(?=\n\s*\d+\s+\d{2}:\d{2}:\d{2}[.,]\d{3}\s*-->|\Z)" + blocks = [] + for m in re.finditer(pattern, content, re.DOTALL): + text = " ".join(line.strip() for line in m.group(4).splitlines() if line.strip()) + start_str = m.group(2).strip() + end_str = m.group(3).strip() + if text: + blocks.append({ + "id": int(m.group(1)), + "timing": f"{start_str} --> {end_str}", + "text": text, + "start_ms": self._ts_to_ms(start_str), + "end_ms": self._ts_to_ms(end_str), + }) + return blocks + + @staticmethod + def _ts_to_ms(ts: str) -> int: + ts = ts.strip().replace(".", ",") + m = re.match(r"(\d+):(\d+):(\d+)[,](\d+)", ts) + if m: + h, mins, s, ms = map(int, m.groups()) + return ((h * 3600 + mins * 60 + s) * 1000) + ms + return 0 + + def _build_system_prompt(self) -> str: + glossary_sample = ", ".join([f"{k} -> {v}" for k, v in list(self.glossary.items())[:35]]) + return ( + "Bạn là chuyên gia dịch thuật video Sneaker, Thời trang Streetwear, Review sản phẩm từ tiếng Trung/Anh sang tiếng Việt tự nhiên, sành điệu, bắt trend Gen Z.\n" + "QUY TẮC BẮT BUỘC:\n" + "1. Dịch từng dòng theo cấu trúc: [N] Câu dịch tiếng Việt hoàn chỉnh.\n" + "2. Giữ nguyên thuật ngữ & thương hiệu tiếng Anh (Nike, Jordan, Yeezy, BAPE, Supreme, Rick Owens, outfit, fit, drip, full box, collab, signature...). \n" + f"3. Bắt buộc áp dụng từ điển chuyên ngành: {glossary_sample}\n" + "4. Dịch ĐẦY ĐỦ Ý NGHĨA trọn vẹn của câu, giữ đủ các từ khoá, KHÔNG bỏ lửng hay cắt cụt mất từ ở cuối câu.\n" + "5. KHÔNG giải thích, CHỈ trả về danh sách các dòng [N] Tiếng Việt." + ) + + def _translate_srt_cloud(self, srt_in: Path, srt_out: Path, source_lang: str = "zh"): + content = srt_in.read_text(encoding="utf-8", errors="ignore") + blocks = self._parse_srt_blocks(content) + + if not blocks: + srt_out.write_text(content, encoding="utf-8") + return + + system_prompt = self._build_system_prompt() + validator = TranslationValidator() + trans_map = {} + + # Phase 1: Batch translation in chunks of 20 + chunk_size = 20 + for i in range(0, len(blocks), chunk_size): + chunk = blocks[i:i + chunk_size] + chunk_result = self._translate_chunk_with_retry(chunk, system_prompt, validator, max_retries=2) + trans_map.update(chunk_result) + + # Phase 2: Retry missing / Chinese-leaked blocks + missing_blocks = [b for b in blocks if (b["id"] not in trans_map or has_chinese(trans_map.get(b["id"], "")))] + if missing_blocks: + self._log(f"🔄 Đang hoàn thiện nốt {len(missing_blocks)} câu dịch còn lại...") + retry_result = self._translate_chunk_with_retry(missing_blocks, system_prompt, validator, max_retries=2) + trans_map.update(retry_result) + + # Phase 3: Google Translate fallback + still_missing = [b for b in blocks if (b["id"] not in trans_map or has_chinese(trans_map.get(b["id"], "")))] + if still_missing: + self._log(f"🌐 {len(still_missing)} câu cần fallback Google Translate...") + google_fb = GoogleTranslateFallback(log_fn=self._log) + google_result = google_fb.translate_blocks(still_missing) + for bid_str, text in google_result.items(): + trans_map[int(bid_str)] = text + + # Phase 4: Post-editing + self._log("✨ Chạy Post-Editor sửa lỗi dịch...") + for b in blocks: + bid = b["id"] + if bid in trans_map: + trans_map[bid] = post_edit_translation(b["text"], trans_map[bid]) + + # Phase 5: Quality Guard + self._log("🛡️ Quality Guard kiểm tra chất lượng bản dịch...") + guard = TranslationQualityGuard(min_score=70) + guard_dict = {str(b["id"]): trans_map.get(b["id"], b["text"]) for b in blocks} + fixed_dict, quality_report = guard.audit_and_fix(blocks, guard_dict) + + for b in blocks: + bid_str = str(b["id"]) + if bid_str in fixed_dict: + trans_map[b["id"]] = fixed_dict[bid_str] + + # Phase 6: Timing compaction + self._log("⏱️ Rút gọn câu dịch cho vừa timeline TTS...") + compact_dict = {str(b["id"]): trans_map.get(b["id"], b["text"]) for b in blocks} + compact_dict = compact_blocks_for_tts(blocks, compact_dict) + for b in blocks: + bid_str = str(b["id"]) + if bid_str in compact_dict: + trans_map[b["id"]] = compact_dict[bid_str] + + # Phase 7: Normalize + Output SRT + out_lines = [] + for b in blocks: + vi_text = trans_map.get(b["id"], b["text"]) + vi_text = self.normalizer.normalize(vi_text) if hasattr(self.normalizer, "normalize") else vi_text + out_lines.append(str(b["id"])) + out_lines.append(b["timing"]) + out_lines.append(vi_text) + out_lines.append("") + + srt_out.write_text("\n".join(out_lines), encoding="utf-8") + self._log(f"✅ Dịch hoàn tất {len(blocks)} câu với 7 bước kiểm tra chất lượng.") + + def _translate_chunk_with_retry( + self, + chunk: List[Dict], + system_prompt: str, + validator: TranslationValidator, + max_retries: int = 2 + ) -> Dict[int, str]: + prompt_lines = [f"[{b['id']}] {b['text']}" for b in chunk] + full_transcript = "\n".join(prompt_lines) + result = {} + + for attempt in range(max_retries + 1): + raw_res = self._direct_25_model_translate(system_prompt, full_transcript) + cleaned_res = re.sub(r".*?", "", raw_res, flags=re.DOTALL).strip() + + attempt_map = {} + for line in cleaned_res.splitlines(): + m = re.match(r"^\s*\[(\d+)\]\s*(.*)$", line.strip()) + if m: + attempt_map[int(m.group(1))] = m.group(2).strip() + + val_dict = {str(b["id"]): attempt_map.get(b["id"], "") for b in chunk if b["id"] in attempt_map} + val_chunk = [b for b in chunk if b["id"] in attempt_map] + + if val_chunk and val_dict: + is_valid, error_msg = validator.validate(val_chunk, val_dict) + if is_valid: + result.update(attempt_map) + return result + else: + for bid, text in attempt_map.items(): + src_block = next((b for b in chunk if b["id"] == bid), None) + if src_block: + ok, _ = validator.validate_single_block(bid, src_block["text"], text) + if ok: + result[bid] = text + else: + result.update(attempt_map) + + return result + + def _direct_25_model_translate(self, system_prompt: str, user_content: str) -> str: + messages = [ + {"role": "system", "content": system_prompt}, + {"role": "user", "content": user_content} + ] + + # 1. Groq + groq_keys = self.asr_engine.groq_keys + for model in ["qwen/qwen3.6-27b", "openai/gpt-oss-120b", "openai/gpt-oss-20b", "groq/compound"]: + for key in groq_keys: + try: + res = requests.post( + "https://api.groq.com/openai/v1/chat/completions", + headers={"Authorization": f"Bearer {key}", "Content-Type": "application/json"}, + json={"model": model, "messages": messages, "temperature": 0.3, "max_tokens": 4096}, + timeout=30 + ) + if res.status_code == 200: + content = res.json()["choices"][0]["message"]["content"].strip() + content = re.sub(r".*?", "", content, flags=re.DOTALL).strip() + if len(content) > 15: + return content + except Exception: + pass + + # 2. Google Gemini + gemini_keys = self.ocr_engine.gemini_keys + for model in ["gemini-3.6-flash", "gemini-flash-latest", "gemini-3.5-flash", "gemini-pro-latest"]: + for key in gemini_keys: + try: + url = f"https://generativelanguage.googleapis.com/v1beta/models/{model}:generateContent?key={key}" + payload = { + "contents": [{"parts": [{"text": f"{system_prompt}\n\n{user_content}"}]}], + "generationConfig": {"temperature": 0.3, "maxOutputTokens": 4096} + } + res = requests.post(url, json=payload, timeout=30) + if res.status_code == 200: + content = res.json()["candidates"][0]["content"]["parts"][0]["text"].strip() + if len(content) > 15: + return content + except Exception: + pass + + # 3. OpenRouter + or_keys = self.ocr_engine.openrouter_keys + for model in ["nvidia/nemotron-3-super-120b-a12b:free", "nvidia/nemotron-3.5-lightning:free", "openai/gpt-oss-20b:free"]: + for key in or_keys: + try: + res = requests.post( + "https://openrouter.ai/api/v1/chat/completions", + headers={"Authorization": f"Bearer {key}", "Content-Type": "application/json", "HTTP-Referer": "https://trungsangviet.local", "X-Title": "TrungSangViet"}, + json={"model": model, "messages": messages, "temperature": 0.3, "max_tokens": 4096}, + timeout=35 + ) + if res.status_code == 200: + content = res.json()["choices"][0]["message"]["content"].strip() + if len(content) > 15: + return content + except Exception: + pass + + return user_content