diff --git "a/app/core/cloud_pipeline.py" "b/app/core/cloud_pipeline.py"
--- "a/app/core/cloud_pipeline.py"
+++ "b/app/core/cloud_pipeline.py"
@@ -1,1208 +1,1208 @@
-"""
-app/core/cloud_pipeline.py
-──────────────────────────
-Pure Cloud-Native Headless Pipeline Coordinator.
-Features:
-1. High-Definition Video Preview Frame Generator using FFmpeg (100% Reliable).
-2. Multi-Model ASR Matrix (Groq Whisper Large V3 / OpenRouter / Gemini).
-3. AI Vision Models OCR with automatic ASR fallback.
-4. Active 2026 AI Translation Models (Groq Qwen 3.6 / GPT-OSS 120B / Gemini Flash / Nemotron 3).
-5. Glossary & Slang Mapping (Fashion, Sneaker, Streetwear, Chinese Slang).
-6. 7-Step Quality Guard & Semantic Validation & Timing Compaction.
-7. Subtitle Editing Hooks (extract_subtitles_only & render_from_subtitles).
-8. ASS Subtitle Engine with TikTok / Shorts Typography & Word Jump Styles.
-9. 100% Solid/Delogo Masking Box to eliminate old Chinese subtitles.
-10. Studio SFX & Audio Ducking Mixer.
-11. SQLite JobManager Integration for state persistence.
-"""
-
-import os
-import re
-import sys
-import json
-import time
-import base64
-import shutil
-import cv2
-import requests
-import subprocess
-from pathlib import Path
-from typing import Optional, Callable, Dict, Any, Tuple, List
-
-from app.core.cloud_asr import CloudASREngine
-from app.core.cloud_ocr import CloudOCREngine
-from app.core.cloud_tts import CloudTTSEngine
-from app.core.vietnamese_text_normalizer import VietnameseTextNormalizer
-from app.core.translation_validator import TranslationValidator
-from app.core.translation_quality_guard import TranslationQualityGuard
-from app.core.translation_post_editor import post_edit_translation
-from app.core.google_translate_fallback import GoogleTranslateFallback
-from app.core.subtitle_compactor import compact_blocks_for_tts
-from app.core.job_manager import JobManager
-from app.core.studio_sfx import generate_sfx_clip
-
-
-def has_chinese(text: str) -> bool:
- """Returns True if string contains CJK Chinese characters."""
- return bool(re.search(r"[\u4e00-\u9fff]", text))
-
-
-class CloudPipeline:
- def __init__(
- self,
- base_dir: Optional[Path] = None,
- log_callback: Optional[Callable[[str], None]] = None,
- progress_callback: Optional[Callable[[int, str], None]] = None,
- ffmpeg_path: str = "ffmpeg"
- ):
- self.base_dir = Path(base_dir or Path(__file__).resolve().parents[2])
- self.output_dir = self.base_dir / "output"
- self.temp_dir = self.base_dir / "temp"
- self.output_dir.mkdir(parents=True, exist_ok=True)
- self.temp_dir.mkdir(parents=True, exist_ok=True)
-
- self.log_fn = log_callback or print
- self.progress_fn = progress_callback or (lambda pct, stage: None)
- self.ffmpeg_path = ffmpeg_path
-
- self.asr_engine = CloudASREngine(log_fn=self._log)
- self.ocr_engine = CloudOCREngine(log_fn=self._log)
- self.tts_engine = CloudTTSEngine(log_fn=self._log, ffmpeg_path=self.ffmpeg_path)
- self.normalizer = VietnameseTextNormalizer()
- self.job_manager = JobManager.instance()
- self.glossary = self._load_glossary()
-
- def _log(self, msg: str):
- self.log_fn(f"[Cloud Pipeline] {msg}")
-
- def _load_glossary(self) -> Dict[str, str]:
- glossary_path = self.base_dir / "glossary.json"
- if glossary_path.exists():
- try:
- data = json.loads(glossary_path.read_text(encoding="utf-8"))
- return data.get("glossary", {})
- except Exception as e:
- self._log(f"⚠️ Lỗi đọc glossary.json: {e}")
- return {}
-
- def calculate_region(
- self,
- video_path: Path,
- region_preset: str = "custom",
- custom_y_pct: float = 75.0,
- custom_h_pct: float = 20.0,
- custom_x_pct: float = 0.0,
- custom_w_pct: float = 100.0
- ) -> Tuple[int, int, int, int]:
- vw, vh = 1920, 1080
- try:
- cap = cv2.VideoCapture(str(video_path))
- w = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH))
- h = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT))
- cap.release()
- if w > 0 and h > 0:
- vw, vh = w, h
- except Exception:
- pass
-
- if region_preset == "bottom_25":
- x = 0
- w = vw
- y = int(vh * 0.72)
- h = int(vh * 0.24)
- elif region_preset == "bottom_15":
- x = 0
- w = vw
- y = int(vh * 0.82)
- h = int(vh * 0.15)
- elif region_preset == "middle_25":
- x = 0
- w = vw
- y = int(vh * 0.38)
- h = int(vh * 0.24)
- elif region_preset == "none":
- return (0, 0, 0, 0)
- elif region_preset == "custom":
- x = int(vw * (custom_x_pct / 100.0))
- w = int(vw * (custom_w_pct / 100.0))
- y = int(vh * (custom_y_pct / 100.0))
- h = int(vh * (custom_h_pct / 100.0))
- else:
- return (0, 0, 0, 0)
-
- x = max(0, min(vw - 2, x))
- y = max(0, min(vh - 2, y))
- w = max(2, min(vw - x, w))
- h = max(2, min(vh - y, h))
- return (x, y, w, h)
-
- def generate_preview_frame(
- self,
- video_path: str,
- region_preset: str = "custom",
- custom_y_pct: float = 75.0,
- custom_h_pct: float = 20.0,
- custom_x_pct: float = 0.0,
- custom_w_pct: float = 100.0,
- sec: float = 2.0,
- return_type: str = "rgb",
- # ── Logo overlay params ──
- logo_enabled: bool = False,
- logo_path: str = "",
- logo_preset: str = "bottom_right",
- logo_scale: float = 15.0,
- logo_opacity: float = 1.0,
- logo_x_pct: float = 80.0,
- logo_y_pct: float = 80.0,
- logo_chromakey: bool = True,
- # ── CTA video overlay params (appears every 5s) ──
- cta_enabled: bool = False,
- cta_path: str = "",
- cta_preset: str = "bottom_center",
- cta_scale: float = 35.0,
- cta_opacity: float = 1.0,
- cta_x_pct: float = 50.0,
- cta_y_pct: float = 85.0,
- cta_interval: float = 5.0,
- cta_duration: float = 2.0,
- cta_chromakey: bool = True
- ):
- """Extracts a frame at timestamp sec and overlays bounding box + logo + CTA preview."""
- v_file = Path(video_path)
- if not v_file.exists():
- return None
-
- raw_frame_path = self.temp_dir / f"raw_frame_{int(time.time()*1000)}.jpg"
- cmd = [
- str(self.ffmpeg_path), "-ss", str(sec), "-y",
- "-i", str(v_file),
- "-frames:v", "1", "-q:v", "2",
- str(raw_frame_path)
- ]
- try:
- subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True)
- frame = cv2.imread(str(raw_frame_path))
- if raw_frame_path.exists():
- raw_frame_path.unlink()
- except Exception:
- frame = None
-
- if frame is None:
- cap = cv2.VideoCapture(str(v_file))
- fps = cap.get(cv2.CAP_PROP_FPS) or 25.0
- target_frame = max(0, int(sec * fps))
- cap.set(cv2.CAP_PROP_POS_FRAMES, target_frame)
- ret, frame = cap.read()
- cap.release()
-
- if frame is None:
- return None
-
- x, y, w, h = self.calculate_region(
- v_file, region_preset, custom_y_pct, custom_h_pct, custom_x_pct, custom_w_pct
- )
-
- if w > 0 and h > 0:
- overlay = frame.copy()
- cv2.rectangle(overlay, (x, y), (x + w, y + h), (0, 0, 255), -1)
- cv2.addWeighted(overlay, 0.35, frame, 0.65, 0, frame)
- cv2.rectangle(frame, (x, y), (x + w, y + h), (0, 255, 255), 3)
-
- label = "VUNG CHE SUB CU & QUET OCR"
- font = cv2.FONT_HERSHEY_SIMPLEX
- font_scale = max(0.5, frame.shape[1] / 1200.0)
- thickness = 2
- (tw, th), _ = cv2.getTextSize(label, font, font_scale, thickness)
-
- label_y = max(th + 10, y - 8)
- cv2.rectangle(frame, (x, label_y - th - 6), (x + tw + 10, label_y + 4), (0, 0, 0), -1)
- cv2.putText(frame, label, (x + 5, label_y), font, font_scale, (0, 255, 255), thickness, cv2.LINE_AA)
-
- # ── Logo overlay preview ──
- if logo_enabled and logo_path and Path(logo_path).exists():
- try:
- from app.core.logo_overlay import load_logo_rgba, overlay_logo_on_frame
- _logo_rgba = load_logo_rgba(logo_path, chromakey=logo_chromakey)
- if _logo_rgba is not None:
- frame = overlay_logo_on_frame(
- frame_bgr=frame,
- logo_rgba=_logo_rgba,
- preset=logo_preset,
- scale_pct=float(logo_scale),
- opacity=float(logo_opacity),
- custom_x_pct=float(logo_x_pct),
- custom_y_pct=float(logo_y_pct),
- margin_pct=2.0
- )
- cv2.putText(frame, f"LOGO:{logo_preset} {logo_scale:.0f}%", (10, 30), cv2.FONT_HERSHEY_SIMPLEX, 0.7, (0, 255, 255), 2, cv2.LINE_AA)
- except Exception as _e:
- self._log(f"⚠️ Logo preview error: {_e}")
-
- # ── CTA video overlay preview (every 5s) ──
- if cta_enabled and cta_path and Path(cta_path).exists():
- try:
- from app.core.logo_overlay import load_cta_frame_rgba, overlay_cta_on_frame, should_show_cta_at_time
- if should_show_cta_at_time(float(sec), float(cta_interval), float(cta_duration)):
- # CTA time loops within CTA video duration
- _cta_time = float(sec) % float(cta_interval)
- # If CTA video is shorter than interval, loop inside
- _cta_rgba = load_cta_frame_rgba(cta_path, cta_time_sec=_cta_time, chromakey=cta_chromakey)
- if _cta_rgba is not None:
- frame = overlay_cta_on_frame(
- frame_bgr=frame,
- cta_rgba=_cta_rgba,
- preset=cta_preset,
- scale_pct=float(cta_scale),
- opacity=float(cta_opacity),
- custom_x_pct=float(cta_x_pct),
- custom_y_pct=float(cta_y_pct),
- margin_pct=2.0
- )
- cv2.putText(frame, f"CTA:{cta_preset} {cta_scale:.0f}% every {cta_interval:.0f}s", (10, 60), cv2.FONT_HERSHEY_SIMPLEX, 0.6, (0, 255, 0), 2, cv2.LINE_AA)
- else:
- cv2.putText(frame, f"CTA: hidden at {sec:.1f}s (interval {cta_interval:.0f}s)", (10, 60), cv2.FONT_HERSHEY_SIMPLEX, 0.5, (0, 165, 255), 1, cv2.LINE_AA)
- except Exception as _e:
- self._log(f"⚠️ CTA preview error: {_e}")
-
- if return_type == "base64":
- _, buffer = cv2.imencode('.jpg', frame, [cv2.IMWRITE_JPEG_QUALITY, 85])
- return base64.b64encode(buffer).decode('utf-8')
-
- return cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)
-
- def extract_subtitles_only(
- self,
- video_path: str,
- mode: str = "asr",
- source_lang: str = "zh",
- region_preset: str = "custom",
- custom_y_pct: float = 75.0,
- custom_h_pct: float = 20.0,
- custom_x_pct: float = 0.0,
- custom_w_pct: float = 100.0
- ) -> Optional[Dict[str, Any]]:
- """
- Runs Stage 0 -> Stage A -> Stage C.
- Returns parsed subtitle blocks (original & translated) for Web Subtitle Editor.
- """
- v_path = Path(video_path)
- if not v_path.exists():
- self._log(f"❌ Video not found: {video_path}")
- return None
-
- video_stem = v_path.stem
- vtd = self.temp_dir / video_stem
- vtd.mkdir(parents=True, exist_ok=True)
-
- self.progress_fn(5, "STAGE_0_PREPARE")
- calc_region = self.calculate_region(
- v_path, region_preset, custom_y_pct, custom_h_pct, custom_x_pct, custom_w_pct
- )
- x, y, w, h = calc_region
-
- # 1. Extract audio
- extracted_audio = vtd / "extracted_audio.wav"
- self._log("⚡ Trích xuất âm thanh từ video gốc...")
- cmd = [
- str(self.ffmpeg_path), "-y", "-i", str(v_path),
- "-vn", "-acodec", "pcm_s16le", "-ar", "16000", "-ac", "1",
- str(extracted_audio)
- ]
- subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True)
-
- # 2. OCR / ASR
- self.progress_fn(25, "STAGE_A_OCR_ASR")
- original_srt = vtd / "original.srt"
- if mode == "ocr":
- self._log("👁️ Quét chữ phụ đề bằng AI Vision Models...")
- ocr_box = (x, y, w, h) if (w > 0 and h > 0) else None
- ok = self.ocr_engine.scan_video_subtitles_to_srt(str(v_path), str(original_srt), blur_region=ocr_box)
- if not ok or not original_srt.exists() or original_srt.stat().st_size < 10:
- self._log("⚠️ OCR không phát hiện chữ -> Tự động chuyển sang Cloud ASR...")
- ok = self.asr_engine.transcribe_audio_to_srt(str(extracted_audio), str(original_srt), source_lang=source_lang)
- else:
- self._log("🎙️ Nhận diện giọng nói bằng Cloud ASR (Groq Whisper Large-v3)...")
- ok = self.asr_engine.transcribe_audio_to_srt(str(extracted_audio), str(original_srt), source_lang=source_lang)
-
- if not ok or not original_srt.exists() or original_srt.stat().st_size < 10:
- raise RuntimeError("Không thể trích xuất phụ đề từ video.")
-
- # 3. Translation
- self.progress_fn(50, "STAGE_C_TRANSLATION")
- self._log("🌐 Dịch thuật bằng AI Translation Matrix...")
- translated_srt = vtd / "translated.srt"
- self._translate_srt_cloud(original_srt, translated_srt, source_lang=source_lang)
-
- orig_blocks = self._parse_srt_blocks(original_srt.read_text(encoding="utf-8", errors="ignore"))
- trans_blocks = self._parse_srt_blocks(translated_srt.read_text(encoding="utf-8", errors="ignore"))
-
- combined = []
- for ob in orig_blocks:
- tb = next((t for t in trans_blocks if t["id"] == ob["id"]), None)
- combined.append({
- "id": ob["id"],
- "timing": ob["timing"],
- "start_ms": ob["start_ms"],
- "end_ms": ob["end_ms"],
- "original_text": ob["text"],
- "vietnamese_text": tb["text"] if tb else ob["text"],
- "sfx": ""
- })
-
- return {
- "video_stem": video_stem,
- "blocks": combined,
- "original_srt": str(original_srt),
- "translated_srt": str(translated_srt)
- }
-
- def render_from_subtitles(
- self,
- video_path: str,
- subtitles_data: List[Dict[str, Any]],
- voice: str = "vi-VN-NamMinhNeural",
- speed: float = 1.0,
- pitch: int = 0,
- volume: int = 100,
- region_preset: str = "custom",
- sub_mask_mode: str = "box",
- sub_style: str = "motion_drip",
- custom_y_pct: float = 75.0,
- custom_h_pct: float = 20.0,
- custom_x_pct: float = 0.0,
- custom_w_pct: float = 100.0,
- ducking_ratio: float = 0.18,
- enable_sfx: bool = True,
- mute_original_audio: bool = False,
- # ── Logo overlay ──
- logo_enabled: bool = False,
- logo_path: str = "",
- logo_preset: str = "bottom_right",
- logo_scale: float = 15.0,
- logo_opacity: float = 1.0,
- logo_x_pct: float = 80.0,
- logo_y_pct: float = 80.0,
- logo_chromakey: bool = True,
- # ── CTA video overlay (every 5s) ──
- cta_enabled: bool = False,
- cta_path: str = "",
- cta_preset: str = "bottom_center",
- cta_scale: float = 35.0,
- cta_opacity: float = 1.0,
- cta_x_pct: float = 50.0,
- cta_y_pct: float = 85.0,
- cta_interval: float = 5.0,
- cta_duration: float = 2.0,
- cta_chromakey: bool = True
- ) -> Optional[str]:
- """Completes TTS synthesis, Ducking audio mix, and video rendering from user-reviewed subtitles."""
- v_path = Path(video_path)
- if not v_path.exists():
- return None
-
- video_stem = v_path.stem
- vtd = self.temp_dir / video_stem
- vtd.mkdir(parents=True, exist_ok=True)
-
- start_time = time.time()
- calc_region = self.calculate_region(
- v_path, region_preset, custom_y_pct, custom_h_pct, custom_x_pct, custom_w_pct
- )
- x, y, w, h = calc_region
-
- # Write edited SRT
- translated_srt = vtd / "translated.srt"
- srt_lines = []
- for item in subtitles_data:
- srt_lines.append(str(item["id"]))
- srt_lines.append(item["timing"])
- vi_text = self.normalizer.normalize(item["vietnamese_text"]) if hasattr(self.normalizer, "normalize") else item["vietnamese_text"]
- srt_lines.append(vi_text)
- srt_lines.append("")
- translated_srt.write_text("\n".join(srt_lines), encoding="utf-8")
-
- # 4. Cloud TTS
- self.progress_fn(65, "STAGE_TTS_DUBBING")
- self._log(f"🎙️ Tạo giọng đọc lồng tiếng ({voice}, speed={speed}x, volume={volume}%)...")
- dubbing_wav = vtd / "dubbing.wav"
- ok_tts = self.tts_engine.synthesize_srt_to_audio(
- str(translated_srt),
- str(dubbing_wav),
- voice=voice,
- speed=speed,
- pitch=pitch,
- volume=volume,
- temp_dir=str(vtd / "tts_segments")
- )
- if not ok_tts or not dubbing_wav.exists():
- raise RuntimeError("Lỗi tạo giọng đọc TTS.")
-
- # 5. SFX & Audio Ducking
- self.progress_fn(80, "STAGE_D_AUDIO_MIX")
- extracted_audio = vtd / "extracted_audio.wav"
- if not extracted_audio.exists():
- cmd = [
- str(self.ffmpeg_path), "-y", "-i", str(v_path),
- "-vn", "-acodec", "pcm_s16le", "-ar", "16000", "-ac", "1",
- str(extracted_audio)
- ]
- subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True)
-
- mixed_audio = vtd / "mixed_final.wav"
- if mute_original_audio:
- self._log("🔇 Tắt hoàn toàn tiếng gốc — chỉ giữ giọng Việt (mute_original_audio=ON)")
- # Dubbing wav đã được time-stretch canvas đúng duration, chỉ cần chuẩn hóa sample-rate
- cmd_mix = [
- str(self.ffmpeg_path), "-y",
- "-i", str(dubbing_wav),
- "-ac", "2", "-ar", "48000",
- str(mixed_audio)
- ]
- subprocess.run(cmd_mix, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True)
- else:
- self._log("🎚️ Trộn nhạc nền (Auto Ducking) + Giọng đọc thuyết minh...")
- filter_complex = (
- f"[0:a]volume={ducking_ratio}[bg];"
- f"[1:a]volume=1.0[dub];"
- f"[bg][dub]amix=inputs=2:duration=longest:dropout_transition=2:normalize=0[aout]"
- )
- cmd_mix = [
- str(self.ffmpeg_path), "-y",
- "-i", str(extracted_audio),
- "-i", str(dubbing_wav),
- "-filter_complex", filter_complex,
- "-map", "[aout]",
- "-ac", "2", "-ar", "48000",
- str(mixed_audio)
- ]
- subprocess.run(cmd_mix, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True)
-
- # 6. Render Video
- self.progress_fn(90, "STAGE_D_RENDER")
- self._log("🎬 Render video hoàn thiện: Xóa sub cũ & Đè sub tiếng Việt...")
- final_output = self.output_dir / f"studio_final_{v_path.name}"
-
- # Probe dimensions with cv2 then ffprobe fallback (handles vertical & cv2-missing envs)
- vw, vh = 1920, 1080
- try:
- cap = cv2.VideoCapture(str(v_path))
- w_tmp = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH)) or 0
- h_tmp = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT)) or 0
- cap.release()
- if w_tmp > 0 and h_tmp > 0:
- vw, vh = w_tmp, h_tmp
- else:
- raise ValueError("cv2 returned 0")
- except Exception:
- try:
- import shutil as _sh
- _ffprobe = _sh.which("ffprobe") or str(Path(self.ffmpeg_path).parent / "ffprobe.exe") if Path(self.ffmpeg_path).exists() else "ffprobe"
- if not Path(_ffprobe).exists():
- _ffprobe = "ffprobe"
- _res = subprocess.run([_ffprobe, "-v", "error", "-select_streams", "v:0", "-show_entries", "stream=width,height", "-of", "json", str(v_path)], capture_output=True, text=True, timeout=5)
- _j = json.loads(_res.stdout or "{}")
- _ws = _j.get("streams", [{}])[0]
- if _ws.get("width") and _ws.get("height"):
- vw, vh = int(_ws["width"]), int(_ws["height"])
- except Exception:
- pass
-
- translated_ass = vtd / "translated.ass"
- self._convert_srt_to_ass(translated_srt, translated_ass, vw, vh, x, y, w, h, sub_style)
-
- ass_escaped = str(translated_ass).replace("\\", "/").replace(":", "\\:")
- sub_filter = f"subtitles='{ass_escaped}'"
-
- vf_filters = []
- if w > 0 and h > 0:
- if sub_mask_mode == "delogo":
- safe_x = max(2, x)
- safe_y = max(2, y)
- safe_w = max(4, min(w, vw - safe_x - 2))
- safe_h = max(4, min(h, vh - safe_y - 2))
- vf_filters.append(f"delogo=x={safe_x}:y={safe_y}:w={safe_w}:h={safe_h}")
- elif sub_mask_mode == "box":
- vf_filters.append(f"drawbox=x={x}:y={y}:w={w}:h={h}:color=black@0.92:t=fill")
-
- # Determine logo & CTA overlay enabled (with fallback to bundled assets)
- _logo_enabled = bool(logo_enabled and logo_path and Path(logo_path).exists())
- if logo_enabled and not _logo_enabled:
- _fb = self.base_dir / "assets" / "logo_ins_drippy4.png"
- if _fb.exists():
- _logo_enabled = True
- logo_path = str(_fb)
- _logo_path = Path(logo_path) if _logo_enabled else None
- _cta_enabled = bool(cta_enabled and cta_path and Path(cta_path).exists())
- if cta_enabled and not _cta_enabled:
- _fb2 = self.base_dir / "assets" / "cta_ins_drippy4.mp4"
- if _fb2.exists():
- _cta_enabled = True
- cta_path = str(_fb2)
- _cta_path = Path(cta_path) if _cta_enabled else None
-
- # If any overlay enabled, we need filter_complex
- if _logo_enabled or _cta_enabled:
- if _logo_enabled:
- self._log(f"🖼️ Overlay logo: {Path(logo_path).name} preset={logo_preset} scale={logo_scale}% opacity={logo_opacity} chromakey={logo_chromakey}")
- if _cta_enabled:
- self._log(f"🎬 Overlay CTA: {Path(cta_path).name} preset={cta_preset} scale={cta_scale}% every {cta_interval}s for {cta_duration}s chromakey={cta_chromakey}")
- # Build filter parts step-by-step. Inputs: 0:video, 1:audio, 2:logo(if), 3:cta(if)
- # Determine indices
- _logo_idx = 2 if _logo_enabled else None
- _cta_idx = None
- if _cta_enabled:
- _cta_idx = 3 if _logo_enabled else 2
- # Video preprocessing (delogo/box) -> [base]
- _filter_parts = []
- if vf_filters and len([f for f in vf_filters if f != sub_filter]) > 0:
- _pre_vf = ",".join([f for f in vf_filters if f != sub_filter])
- _filter_parts.append(f"[0:v]{_pre_vf}[base]")
- _cur = "base"
- else:
- _filter_parts.append("[0:v]null[base]")
- _cur = "base"
- # Logo overlay
- if _logo_enabled:
- _logo_w_orig, _logo_h_orig = 2400, 1792
- try:
- _t = cv2.imread(str(_logo_path), cv2.IMREAD_UNCHANGED)
- if _t is not None:
- _logo_h_orig, _logo_w_orig = _t.shape[:2]
- except Exception:
- pass
- _target_w = vw * float(logo_scale) / 100.0
- _sf = max(0.02, min(0.5, _target_w / float(max(1, _logo_w_orig))))
- _logo_vf_parts = []
- if logo_chromakey:
- _logo_vf_parts.append("colorkey=0x00FF00:0.3:0.1")
- _logo_vf_parts.append("format=rgba")
- _logo_vf_parts.append(f"scale=iw*{_sf:.4f}:ih*{_sf:.4f}:flags=lanczos")
- if float(logo_opacity) < 0.99:
- _logo_vf_parts.append(f"colorchannelmixer=aa={float(logo_opacity):.2f}")
- _logo_vf = ",".join(_logo_vf_parts)
- _m = 2.0
- if logo_preset == "top_left":
- _lx, _ly = f"W*{_m/100:.3f}", f"H*{_m/100:.3f}"
- elif logo_preset == "top_right":
- _lx, _ly = f"W-w-W*{_m/100:.3f}", f"H*{_m/100:.3f}"
- elif logo_preset == "bottom_left":
- _lx, _ly = f"W*{_m/100:.3f}", f"H-h-H*{_m/100:.3f}"
- elif logo_preset == "bottom_right":
- _lx, _ly = f"W-w-W*{_m/100:.3f}", f"H-h-H*{_m/100:.3f}"
- elif logo_preset == "center":
- _lx, _ly = "(W-w)/2", "(H-h)/2"
- elif logo_preset == "top_center":
- _lx, _ly = "(W-w)/2", f"H*{_m/100:.3f}"
- elif logo_preset == "bottom_center":
- _lx, _ly = "(W-w)/2", f"H-h-H*{_m/100:.3f}"
- elif logo_preset == "custom":
- _lx, _ly = f"W*{float(logo_x_pct)/100:.4f}", f"H*{float(logo_y_pct)/100:.4f}"
- else:
- _lx, _ly = f"W-w-W*{_m/100:.3f}", f"H-h-H*{_m/100:.3f}"
- _filter_parts.append(f"[{_logo_idx}:v]{_logo_vf}[logo]")
- _next = "with_logo" if _cta_enabled or sub_filter else "v"
- _filter_parts.append(f"[{_cur}][logo]overlay={_lx}:{_ly}:format=rgb[{_next}]")
- _cur = _next
- # CTA overlay (periodic every interval)
- if _cta_enabled:
- # Probe CTA size
- _cta_w_orig, _cta_h_orig = 1080, 1920
- try:
- _cap = cv2.VideoCapture(str(_cta_path))
- _cta_w_orig = int(_cap.get(cv2.CAP_PROP_FRAME_WIDTH)) or _cta_w_orig
- _cta_h_orig = int(_cap.get(cv2.CAP_PROP_FRAME_HEIGHT)) or _cta_h_orig
- _cap.release()
- except Exception:
- pass
- # TikTok CTA bump: enforce minimum 42% width for mobile visibility
- effective_cta_scale = max(float(cta_scale), 42.0)
- _cta_target_w = vw * effective_cta_scale / 100.0
- _cta_sf = max(0.05, min(1.0, _cta_target_w / float(max(1, _cta_w_orig))))
- # Clamp height to 80% of video height to avoid overflow (CTA vertical on horizontal video)
- _cta_target_h = _cta_h_orig * _cta_sf
- _max_h = vh * 0.90
- if _cta_target_h > _max_h:
- _cta_sf = _max_h / float(max(1, _cta_h_orig))
- _cta_target_w = _cta_w_orig * _cta_sf
- _cta_vf_parts = []
- if cta_chromakey:
- _cta_vf_parts.append("colorkey=0x00FF00:0.3:0.1")
- _cta_vf_parts.append("format=rgba")
- _cta_vf_parts.append(f"scale=iw*{_cta_sf:.4f}:ih*{_cta_sf:.4f}:flags=lanczos")
- if float(cta_opacity) < 0.99:
- _cta_vf_parts.append(f"colorchannelmixer=aa={float(cta_opacity):.2f}")
- _cta_vf = ",".join(_cta_vf_parts)
- _m2 = 2.0
- if cta_preset == "top_left":
- _cx, _cy = f"W*{_m2/100:.3f}", f"H*{_m2/100:.3f}"
- elif cta_preset == "top_right":
- _cx, _cy = f"W-w-W*{_m2/100:.3f}", f"H*{_m2/100:.3f}"
- elif cta_preset == "bottom_left":
- _cx, _cy = f"W*{_m2/100:.3f}", f"H-h-H*{_m2/100:.3f}"
- elif cta_preset == "bottom_right":
- _cx, _cy = f"W-w-W*{_m2/100:.3f}", f"H-h-H*{_m2/100:.3f}"
- elif cta_preset == "center":
- _cx, _cy = "(W-w)/2", "(H-h)/2"
- elif cta_preset == "top_center":
- _cx, _cy = "(W-w)/2", f"H*{_m2/100:.3f}"
- elif cta_preset == "bottom_center":
- _cx, _cy = "(W-w)/2", f"H-h-H*{_m2/100:.3f}"
- elif cta_preset == "custom":
- _cx, _cy = f"W*{float(cta_x_pct)/100:.4f}", f"H*{float(cta_y_pct)/100:.4f}"
- else:
- _cx, _cy = "(W-w)/2", f"H-h-H*{_m2/100:.3f}"
- _enable = f"lt(mod(t\\,{float(cta_interval)})\\,{float(cta_duration)})"
- _filter_parts.append(f"[{_cta_idx}:v]{_cta_vf}[cta]")
- _next2 = "v" if not sub_filter else "with_cta"
- # Use escaped comma for FFmpeg enable expression
- _filter_parts.append(f"[{_cur}][cta]overlay={_cx}:{_cy}:format=rgb:enable='{_enable}'[{_next2}]")
- _cur = _next2
- # TikTok polish: CFR 30 + even dims + light sharpen BEFORE subtitles (keeps text razor sharp)
- tiktok_polish = "fps=30:round=near,scale=trunc(iw/2)*2:trunc(ih/2)*2:flags=lanczos+accurate_rnd+full_chroma_int:sws_dither=ed,unsharp=3:3:0.35:3:3:0.0"
- if sub_filter:
- _filter_parts.append(f"[{_cur}]{tiktok_polish}[polished]")
- _cur = "polished"
- _filter_parts.append(f"[{_cur}]{sub_filter}[v]")
- _cur = "v"
- else:
- _filter_parts.append(f"[{_cur}]{tiktok_polish}[v]")
- _cur = "v"
- _filter_complex = ";".join(_filter_parts)
- # Build ffmpeg inputs: video, audio, logo(if), cta(if) - add -shortest when looped overlays present
- cmd_inputs = [str(self.ffmpeg_path), "-y", "-fflags", "+genpts", "-avoid_negative_ts", "make_zero", "-i", str(v_path), "-i", str(mixed_audio)]
- if _logo_enabled:
- cmd_inputs += ["-loop", "1", "-i", str(_logo_path)]
- if _cta_enabled:
- cmd_inputs += ["-stream_loop", "999", "-i", str(_cta_path)]
- _extra = ["-shortest"] if (_logo_enabled or _cta_enabled) else []
- # TikTok spec: High Profile yuv420p, 30 CFR, 48k AAC, faststart, accurate sync
- tiktok_video_args = ["-r", "30", "-c:v", "libx264", "-preset", "medium", "-crf", "18", "-profile:v", "high", "-level", "4.1", "-pix_fmt", "yuv420p", "-g", "60", "-keyint_min", "30", "-sc_threshold", "0", "-x264-params", "ref=4:bframes=2:me=hex:subme=7:psy=1:psy-rd=0.8:aq-mode=2", "-colorspace", "bt709", "-color_primaries", "bt709", "-color_trc", "bt709", "-color_range", "tv"]
- tiktok_audio_args = ["-c:a", "aac", "-profile:a", "aac_low", "-ar", "48000", "-ac", "2", "-b:a", "192k", "-af", "aresample=async=1:min_hard_comp=0.100000:first_pts=0"]
- tiktok_mux_args = ["-movflags", "+faststart", "-fflags", "+genpts", "-max_interleave_delta", "100M", "-vsync", "cfr", "-fps_mode", "cfr"]
- cmd_render = cmd_inputs + ["-filter_complex", _filter_complex, "-map", "[v]", "-map", "1:a:0"] + _extra + tiktok_video_args + tiktok_audio_args + tiktok_mux_args + ["-shortest", str(final_output)]
- res = subprocess.run(cmd_render, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True)
- if res.returncode != 0:
- self._log(f"⚠️ Overlay render failed ({res.stderr[:300]}), fallback to TikTok-spec normal render...")
- # Rebuild VF with TikTok polish inserted before subs (same as non-overlay branch)
- tiktok_polish_fb = "fps=30:round=near,scale=trunc(iw/2)*2:trunc(ih/2)*2:flags=lanczos+accurate_rnd+full_chroma_int:sws_dither=ed,unsharp=3:3:0.35:3:3:0.0"
- vf_without_sub_fb = [f for f in vf_filters if f != sub_filter]
- ordered_fb = []
- if vf_without_sub_fb:
- ordered_fb.extend(vf_without_sub_fb)
- ordered_fb.append(tiktok_polish_fb)
- ordered_fb.append(sub_filter)
- final_vf = ",".join(ordered_fb)
- cmd_render = [
- str(self.ffmpeg_path), "-y",
- "-fflags", "+genpts", "-avoid_negative_ts", "make_zero",
- "-i", str(v_path),
- "-i", str(mixed_audio),
- "-vf", final_vf,
- "-map", "0:v:0", "-map", "1:a:0",
- "-r", "30",
- "-c:v", "libx264", "-preset", "medium", "-crf", "18",
- "-profile:v", "high", "-level", "4.1", "-pix_fmt", "yuv420p",
- "-g", "60", "-keyint_min", "30", "-sc_threshold", "0",
- "-x264-params", "ref=4:bframes=2:me=hex:subme=7:psy=1:psy-rd=0.8:aq-mode=2",
- "-colorspace", "bt709", "-color_primaries", "bt709", "-color_trc", "bt709", "-color_range", "tv",
- "-c:a", "aac", "-profile:a", "aac_low", "-ar", "48000", "-ac", "2", "-b:a", "192k",
- "-af", "aresample=async=1:min_hard_comp=0.100000:first_pts=0",
- "-movflags", "+faststart",
- "-fflags", "+genpts", "-max_interleave_delta", "100M",
- "-vsync", "cfr", "-fps_mode", "cfr",
- "-shortest",
- str(final_output)
- ]
- res = subprocess.run(cmd_render, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True)
- if res.returncode != 0:
- self._log(f"⚠️ Fallback render không mask: {res.stderr[:120]}")
- fallback_cmd = [
- str(self.ffmpeg_path), "-y",
- "-fflags", "+genpts",
- "-i", str(v_path),
- "-i", str(mixed_audio),
- "-vf", f"{tiktok_polish_fb},{sub_filter}",
- "-map", "0:v:0", "-map", "1:a:0",
- "-r", "30",
- "-c:v", "libx264", "-preset", "medium", "-crf", "18",
- "-profile:v", "high", "-pix_fmt", "yuv420p",
- "-c:a", "aac", "-profile:a", "aac_low", "-ar", "48000", "-ac", "2", "-b:a", "192k",
- "-movflags", "+faststart",
- "-shortest",
- str(final_output)
- ]
- subprocess.run(fallback_cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True)
- else:
- # TikTok polish inserted before subtitles for sharpness + CFR + even dims
- tiktok_polish = "fps=30:round=near,scale=trunc(iw/2)*2:trunc(ih/2)*2:flags=lanczos+accurate_rnd+full_chroma_int:sws_dither=ed,unsharp=3:3:0.35:3:3:0.0"
- # Order: mask (delogo/box) -> polish -> subtitles
- vf_without_sub = [f for f in vf_filters if f != sub_filter]
- # Rebuild vf in TikTok-optimal order
- ordered_vf = []
- if vf_without_sub:
- ordered_vf.extend(vf_without_sub)
- ordered_vf.append(tiktok_polish)
- ordered_vf.append(sub_filter)
- final_vf = ",".join(ordered_vf)
-
- cmd_render = [
- str(self.ffmpeg_path), "-y",
- "-fflags", "+genpts", "-avoid_negative_ts", "make_zero",
- "-i", str(v_path),
- "-i", str(mixed_audio),
- "-vf", final_vf,
- "-map", "0:v:0", "-map", "1:a:0",
- "-r", "30",
- "-c:v", "libx264", "-preset", "medium", "-crf", "18",
- "-profile:v", "high", "-level", "4.1", "-pix_fmt", "yuv420p",
- "-g", "60", "-keyint_min", "30", "-sc_threshold", "0",
- "-x264-params", "ref=4:bframes=2:me=hex:subme=7:psy=1:psy-rd=0.8:aq-mode=2",
- "-colorspace", "bt709", "-color_primaries", "bt709", "-color_trc", "bt709", "-color_range", "tv",
- "-c:a", "aac", "-profile:a", "aac_low", "-ar", "48000", "-ac", "2", "-b:a", "192k",
- "-af", "aresample=async=1:min_hard_comp=0.100000:first_pts=0",
- "-movflags", "+faststart",
- "-fflags", "+genpts", "-max_interleave_delta", "100M",
- "-vsync", "cfr", "-fps_mode", "cfr",
- "-shortest",
- str(final_output)
- ]
- res = subprocess.run(cmd_render, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True)
- if res.returncode != 0:
- self._log(f"⚠️ Fallback render không mask: {res.stderr[:120]} | retry with minimal TikTok spec")
- # Minimal fallback: at least ensure TikTok audio/video spec
- fallback_cmd = [
- str(self.ffmpeg_path), "-y",
- "-fflags", "+genpts",
- "-i", str(v_path),
- "-i", str(mixed_audio),
- "-vf", f"{tiktok_polish},{sub_filter}",
- "-map", "0:v:0", "-map", "1:a:0",
- "-r", "30",
- "-c:v", "libx264", "-preset", "medium", "-crf", "18",
- "-profile:v", "high", "-pix_fmt", "yuv420p",
- "-c:a", "aac", "-profile:a", "aac_low", "-ar", "48000", "-ac", "2", "-b:a", "192k",
- "-movflags", "+faststart",
- "-shortest",
- str(final_output)
- ]
- subprocess.run(fallback_cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True)
-
- total_elapsed = time.time() - start_time
- self.progress_fn(100, "DONE")
- self._log(f"🎉 HOÀN THÀNH XUẤT SẮC TRONG {total_elapsed:.1f}s!")
- self._log(f"📁 Video đã lưu tại: {final_output}")
- return str(final_output)
-
- def run_video(
- self,
- video_path: str,
- mode: str = "asr",
- source_lang: str = "zh",
- voice: str = "vi-VN-NamMinhNeural",
- speed: float = 1.0,
- pitch: int = 0,
- volume: int = 100,
- region_preset: str = "custom",
- sub_mask_mode: str = "box",
- sub_style: str = "motion_drip",
- custom_y_pct: float = 75.0,
- custom_h_pct: float = 20.0,
- custom_x_pct: float = 0.0,
- custom_w_pct: float = 100.0,
- ducking_ratio: float = 0.18,
- enable_sfx: bool = True,
- mute_original_audio: bool = False,
- logo_enabled: bool = False,
- logo_path: str = "",
- logo_preset: str = "bottom_right",
- logo_scale: float = 15.0,
- logo_opacity: float = 1.0,
- logo_x_pct: float = 80.0,
- logo_y_pct: float = 80.0,
- logo_chromakey: bool = True,
- cta_enabled: bool = False,
- cta_path: str = "",
- cta_preset: str = "bottom_center",
- cta_scale: float = 35.0,
- cta_opacity: float = 1.0,
- cta_x_pct: float = 50.0,
- cta_y_pct: float = 85.0,
- cta_interval: float = 5.0,
- cta_duration: float = 2.0,
- cta_chromakey: bool = True
- ) -> Optional[str]:
- """Full 1-Click End-to-End Automatic Dubbing Pipeline."""
- v_path = Path(video_path)
- if not v_path.exists():
- self._log(f"❌ Video not found: {video_path}")
- return None
-
- # Register into SQLite Job DB
- try:
- self.job_manager.register_job(str(v_path))
- except Exception:
- pass
-
- try:
- # 1. Extract subtitles
- subs_res = self.extract_subtitles_only(
- video_path=video_path,
- mode=mode,
- source_lang=source_lang,
- region_preset=region_preset,
- custom_y_pct=custom_y_pct,
- custom_h_pct=custom_h_pct,
- custom_x_pct=custom_x_pct,
- custom_w_pct=custom_w_pct
- )
- if not subs_res:
- return None
-
- # 2. Render from subtitles
- return self.render_from_subtitles(
- video_path=video_path,
- subtitles_data=subs_res["blocks"],
- voice=voice,
- speed=speed,
- pitch=pitch,
- volume=volume,
- region_preset=region_preset,
- sub_mask_mode=sub_mask_mode,
- sub_style=sub_style,
- custom_y_pct=custom_y_pct,
- custom_h_pct=custom_h_pct,
- custom_x_pct=custom_x_pct,
- custom_w_pct=custom_w_pct,
- ducking_ratio=ducking_ratio,
- enable_sfx=enable_sfx,
- mute_original_audio=mute_original_audio,
- logo_enabled=logo_enabled,
- logo_path=logo_path,
- logo_preset=logo_preset,
- logo_scale=logo_scale,
- logo_opacity=logo_opacity,
- logo_x_pct=logo_x_pct,
- logo_y_pct=logo_y_pct,
- logo_chromakey=logo_chromakey,
- cta_enabled=cta_enabled,
- cta_path=cta_path,
- cta_preset=cta_preset,
- cta_scale=cta_scale,
- cta_opacity=cta_opacity,
- cta_x_pct=cta_x_pct,
- cta_y_pct=cta_y_pct,
- cta_interval=cta_interval,
- cta_duration=cta_duration,
- cta_chromakey=cta_chromakey
- )
- except Exception as e:
- self._log(f"❌ LỖI PIPELINE: {str(e)}")
- self.progress_fn(0, "FAILED")
- return None
-
- def _convert_srt_to_ass(
- self,
- srt_path: Path,
- ass_path: Path,
- vw: int,
- vh: int,
- box_x: int,
- box_y: int,
- box_w: int,
- box_h: int,
- sub_style: str = "motion_drip"
- ):
- """Converts SRT to styled ASS format for TikTok/Shorts typography."""
- content = srt_path.read_text(encoding="utf-8", errors="ignore")
-
- # TikTok safe zone: keep subtitle inside 82% height, 7% bottom margin, larger font for mobile legibility
- safe_margin_v = max(42, int(vh * 0.07))
- raw_margin_v = int(vh - (box_y + box_h * 0.75)) if box_h > 0 else safe_margin_v
- margin_v = max(safe_margin_v, raw_margin_v)
- # Clamp to avoid bottom UI overlap
- max_bottom = int(vh * 0.12)
- if margin_v < max_bottom:
- margin_v = max_bottom
- # Larger font for TikTok: vertical 56-72, horizontal 48-64
- if box_h > 0:
- base = int(box_h * 0.52)
- if vh > vw: # vertical TikTok
- font_size = max(42, min(72, base))
- else:
- font_size = max(38, min(64, base))
- else:
- font_size = 58 if vh > vw else 50
-
- # Typography Color & Outline palettes - TikTok optimized for high compression
- if sub_style == "motion_drip" or sub_style == "yellow_bold":
- font_color = "&H0000FFFF" # Bright Yellow
- outline_color = "&H00000000" # Pure Black
- back_color = "&H99000000"
- outline = 5.0
- shadow = 2.2
- elif sub_style == "neon_cyan":
- font_color = "&H00FFFF00" # Cyan
- outline_color = "&H00000000"
- back_color = "&H99000000"
- outline = 5.0
- shadow = 2.2
- elif sub_style == "capsule_tag":
- font_color = "&H00FFFFFF" # White
- outline_color = "&H00111111"
- back_color = "&HBB000000"
- outline = 3.2
- shadow = 0.0
- else: # white_bold
- font_color = "&H00FFFFFF" # Crisp White
- outline_color = "&H00000000"
- back_color = "&H99000000"
- outline = 4.8
- shadow = 2.0
-
- ass_header = f"""[Script Info]
-ScriptType: v4.00+
-PlayResX: {vw}
-PlayResY: {vh}
-ScaledBorderAndShadow: yes
-
-[V4+ Styles]
-Format: Name, Fontname, Fontsize, PrimaryColour, SecondaryColour, OutlineColour, BackColour, Bold, Italic, Underline, StrikeOut, ScaleX, ScaleY, Spacing, Angle, BorderStyle, Outline, Shadow, Alignment, MarginL, MarginR, MarginV, Encoding
-Style: Default,DejaVu Sans,{font_size},{font_color},&H000000FF,{outline_color},{back_color},1,0,0,0,100,100,0,0,1,{outline},{shadow},2,40,40,{margin_v},1
-
-[Events]
-Format: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text
-"""
-
- events = []
- pattern = r"(\d+)\s+(\d{2}:\d{2}:\d{2}[.,]\d{3})\s*-->\s*(\d{2}:\d{2}:\d{2}[.,]\d{3})\s*\n(.*?)(?=\n\s*\d+\s+\d{2}:\d{2}:\d{2}[.,]\d{3}\s*-->|\Z)"
- for m in re.finditer(pattern, content, re.DOTALL):
- start_str = m.group(2).strip().replace(",", ".")
- end_str = m.group(3).strip().replace(",", ".")
-
- s_parts = start_str.split(":")
- e_parts = end_str.split(":")
- s_ass = f"{int(s_parts[0])}:{s_parts[1]}:{float(s_parts[2]):05.2f}"
- e_ass = f"{int(e_parts[0])}:{e_parts[1]}:{float(e_parts[2]):05.2f}"
-
- text = " ".join(line.strip() for line in m.group(4).splitlines() if line.strip())
- if text:
- events.append(f"Dialogue: 0,{s_ass},{e_ass},Default,,0,0,0,,{text}")
-
- ass_path.write_text(ass_header + "\n".join(events), encoding="utf-8")
-
- def _parse_srt_blocks(self, content: str) -> List[Dict]:
- pattern = r"(\d+)\s+(\d{2}:\d{2}:\d{2}[.,]\d{3})\s*-->\s*(\d{2}:\d{2}:\d{2}[.,]\d{3})\s*\n(.*?)(?=\n\s*\d+\s+\d{2}:\d{2}:\d{2}[.,]\d{3}\s*-->|\Z)"
- blocks = []
- for m in re.finditer(pattern, content, re.DOTALL):
- text = " ".join(line.strip() for line in m.group(4).splitlines() if line.strip())
- start_str = m.group(2).strip()
- end_str = m.group(3).strip()
- if text:
- blocks.append({
- "id": int(m.group(1)),
- "timing": f"{start_str} --> {end_str}",
- "text": text,
- "start_ms": self._ts_to_ms(start_str),
- "end_ms": self._ts_to_ms(end_str),
- })
- return blocks
-
- @staticmethod
- def _ts_to_ms(ts: str) -> int:
- ts = ts.strip().replace(".", ",")
- m = re.match(r"(\d+):(\d+):(\d+)[,](\d+)", ts)
- if m:
- h, mins, s, ms = map(int, m.groups())
- return ((h * 3600 + mins * 60 + s) * 1000) + ms
- return 0
-
- def _build_system_prompt(self) -> str:
- glossary_sample = ", ".join([f"{k} -> {v}" for k, v in list(self.glossary.items())[:35]])
- return (
- "Bạn là chuyên gia dịch thuật video Sneaker, Thời trang Streetwear, Review sản phẩm từ tiếng Trung/Anh sang tiếng Việt tự nhiên, sành điệu, bắt trend Gen Z.\n"
- "QUY TẮC BẮT BUỘC:\n"
- "1. Dịch từng dòng theo cấu trúc: [N] Câu dịch tiếng Việt hoàn chỉnh.\n"
- "2. Giữ nguyên thuật ngữ & thương hiệu tiếng Anh (Nike, Jordan, Yeezy, BAPE, Supreme, Rick Owens, outfit, fit, drip, full box, collab, signature...). \n"
- f"3. Bắt buộc áp dụng từ điển chuyên ngành: {glossary_sample}\n"
- "4. Dịch ĐẦY ĐỦ Ý NGHĨA trọn vẹn của câu, giữ đủ các từ khoá, KHÔNG bỏ lửng hay cắt cụt mất từ ở cuối câu.\n"
- "5. KHÔNG giải thích, CHỈ trả về danh sách các dòng [N] Tiếng Việt."
- )
-
- def _translate_srt_cloud(self, srt_in: Path, srt_out: Path, source_lang: str = "zh"):
- content = srt_in.read_text(encoding="utf-8", errors="ignore")
- blocks = self._parse_srt_blocks(content)
-
- if not blocks:
- srt_out.write_text(content, encoding="utf-8")
- return
-
- system_prompt = self._build_system_prompt()
- validator = TranslationValidator()
- trans_map = {}
-
- # Phase 1: Batch translation in chunks of 20
- chunk_size = 20
- for i in range(0, len(blocks), chunk_size):
- chunk = blocks[i:i + chunk_size]
- chunk_result = self._translate_chunk_with_retry(chunk, system_prompt, validator, max_retries=2)
- trans_map.update(chunk_result)
-
- # Phase 2: Retry missing / Chinese-leaked blocks
- missing_blocks = [b for b in blocks if (b["id"] not in trans_map or has_chinese(trans_map.get(b["id"], "")))]
- if missing_blocks:
- self._log(f"🔄 Đang hoàn thiện nốt {len(missing_blocks)} câu dịch còn lại...")
- retry_result = self._translate_chunk_with_retry(missing_blocks, system_prompt, validator, max_retries=2)
- trans_map.update(retry_result)
-
- # Phase 3: Google Translate fallback
- still_missing = [b for b in blocks if (b["id"] not in trans_map or has_chinese(trans_map.get(b["id"], "")))]
- if still_missing:
- self._log(f"🌐 {len(still_missing)} câu cần fallback Google Translate...")
- google_fb = GoogleTranslateFallback(log_fn=self._log)
- google_result = google_fb.translate_blocks(still_missing)
- for bid_str, text in google_result.items():
- trans_map[int(bid_str)] = text
-
- # Phase 4: Post-editing
- self._log("✨ Chạy Post-Editor sửa lỗi dịch...")
- for b in blocks:
- bid = b["id"]
- if bid in trans_map:
- trans_map[bid] = post_edit_translation(b["text"], trans_map[bid])
-
- # Phase 5: Quality Guard
- self._log("🛡️ Quality Guard kiểm tra chất lượng bản dịch...")
- guard = TranslationQualityGuard(min_score=70)
- guard_dict = {str(b["id"]): trans_map.get(b["id"], b["text"]) for b in blocks}
- fixed_dict, quality_report = guard.audit_and_fix(blocks, guard_dict)
-
- for b in blocks:
- bid_str = str(b["id"])
- if bid_str in fixed_dict:
- trans_map[b["id"]] = fixed_dict[bid_str]
-
- # Phase 6: Timing compaction
- self._log("⏱️ Rút gọn câu dịch cho vừa timeline TTS...")
- compact_dict = {str(b["id"]): trans_map.get(b["id"], b["text"]) for b in blocks}
- compact_dict = compact_blocks_for_tts(blocks, compact_dict)
- for b in blocks:
- bid_str = str(b["id"])
- if bid_str in compact_dict:
- trans_map[b["id"]] = compact_dict[bid_str]
-
- # Phase 7: Normalize + Output SRT
- out_lines = []
- for b in blocks:
- vi_text = trans_map.get(b["id"], b["text"])
- vi_text = self.normalizer.normalize(vi_text) if hasattr(self.normalizer, "normalize") else vi_text
- out_lines.append(str(b["id"]))
- out_lines.append(b["timing"])
- out_lines.append(vi_text)
- out_lines.append("")
-
- srt_out.write_text("\n".join(out_lines), encoding="utf-8")
- self._log(f"✅ Dịch hoàn tất {len(blocks)} câu với 7 bước kiểm tra chất lượng.")
-
- def _translate_chunk_with_retry(
- self,
- chunk: List[Dict],
- system_prompt: str,
- validator: TranslationValidator,
- max_retries: int = 2
- ) -> Dict[int, str]:
- prompt_lines = [f"[{b['id']}] {b['text']}" for b in chunk]
- full_transcript = "\n".join(prompt_lines)
- result = {}
-
- for attempt in range(max_retries + 1):
- raw_res = self._direct_25_model_translate(system_prompt, full_transcript)
- cleaned_res = re.sub(r".*?", "", raw_res, flags=re.DOTALL).strip()
-
- attempt_map = {}
- for line in cleaned_res.splitlines():
- m = re.match(r"^\s*\[(\d+)\]\s*(.*)$", line.strip())
- if m:
- attempt_map[int(m.group(1))] = m.group(2).strip()
-
- val_dict = {str(b["id"]): attempt_map.get(b["id"], "") for b in chunk if b["id"] in attempt_map}
- val_chunk = [b for b in chunk if b["id"] in attempt_map]
-
- if val_chunk and val_dict:
- is_valid, error_msg = validator.validate(val_chunk, val_dict)
- if is_valid:
- result.update(attempt_map)
- return result
- else:
- for bid, text in attempt_map.items():
- src_block = next((b for b in chunk if b["id"] == bid), None)
- if src_block:
- ok, _ = validator.validate_single_block(bid, src_block["text"], text)
- if ok:
- result[bid] = text
- else:
- result.update(attempt_map)
-
- return result
-
- def _direct_25_model_translate(self, system_prompt: str, user_content: str) -> str:
- messages = [
- {"role": "system", "content": system_prompt},
- {"role": "user", "content": user_content}
- ]
-
- # 1. Groq
- groq_keys = self.asr_engine.groq_keys
- for model in ["qwen/qwen3.6-27b", "openai/gpt-oss-120b", "openai/gpt-oss-20b", "groq/compound"]:
- for key in groq_keys:
- try:
- res = requests.post(
- "https://api.groq.com/openai/v1/chat/completions",
- headers={"Authorization": f"Bearer {key}", "Content-Type": "application/json"},
- json={"model": model, "messages": messages, "temperature": 0.3, "max_tokens": 4096},
- timeout=30
- )
- if res.status_code == 200:
- content = res.json()["choices"][0]["message"]["content"].strip()
- content = re.sub(r".*?", "", content, flags=re.DOTALL).strip()
- if len(content) > 15:
- return content
- except Exception:
- pass
-
- # 2. Google Gemini
- gemini_keys = self.ocr_engine.gemini_keys
- for model in ["gemini-3.6-flash", "gemini-flash-latest", "gemini-3.5-flash", "gemini-pro-latest"]:
- for key in gemini_keys:
- try:
- url = f"https://generativelanguage.googleapis.com/v1beta/models/{model}:generateContent?key={key}"
- payload = {
- "contents": [{"parts": [{"text": f"{system_prompt}\n\n{user_content}"}]}],
- "generationConfig": {"temperature": 0.3, "maxOutputTokens": 4096}
- }
- res = requests.post(url, json=payload, timeout=30)
- if res.status_code == 200:
- content = res.json()["candidates"][0]["content"]["parts"][0]["text"].strip()
- if len(content) > 15:
- return content
- except Exception:
- pass
-
- # 3. OpenRouter
- or_keys = self.ocr_engine.openrouter_keys
- for model in ["nvidia/nemotron-3-super-120b-a12b:free", "nvidia/nemotron-3.5-lightning:free", "openai/gpt-oss-20b:free"]:
- for key in or_keys:
- try:
- res = requests.post(
- "https://openrouter.ai/api/v1/chat/completions",
- headers={"Authorization": f"Bearer {key}", "Content-Type": "application/json", "HTTP-Referer": "https://trungsangviet.local", "X-Title": "TrungSangViet"},
- json={"model": model, "messages": messages, "temperature": 0.3, "max_tokens": 4096},
- timeout=35
- )
- if res.status_code == 200:
- content = res.json()["choices"][0]["message"]["content"].strip()
- if len(content) > 15:
- return content
- except Exception:
- pass
-
- return user_content
+"""
+app/core/cloud_pipeline.py
+──────────────────────────
+Pure Cloud-Native Headless Pipeline Coordinator.
+Features:
+1. High-Definition Video Preview Frame Generator using FFmpeg (100% Reliable).
+2. Multi-Model ASR Matrix (Groq Whisper Large V3 / OpenRouter / Gemini).
+3. AI Vision Models OCR with automatic ASR fallback.
+4. Active 2026 AI Translation Models (Groq Qwen 3.6 / GPT-OSS 120B / Gemini Flash / Nemotron 3).
+5. Glossary & Slang Mapping (Fashion, Sneaker, Streetwear, Chinese Slang).
+6. 7-Step Quality Guard & Semantic Validation & Timing Compaction.
+7. Subtitle Editing Hooks (extract_subtitles_only & render_from_subtitles).
+8. ASS Subtitle Engine with TikTok / Shorts Typography & Word Jump Styles.
+9. 100% Solid/Delogo Masking Box to eliminate old Chinese subtitles.
+10. Studio SFX & Audio Ducking Mixer.
+11. SQLite JobManager Integration for state persistence.
+"""
+
+import os
+import re
+import sys
+import json
+import time
+import base64
+import shutil
+import cv2
+import requests
+import subprocess
+from pathlib import Path
+from typing import Optional, Callable, Dict, Any, Tuple, List
+
+from app.core.cloud_asr import CloudASREngine
+from app.core.cloud_ocr import CloudOCREngine
+from app.core.cloud_tts import CloudTTSEngine
+from app.core.vietnamese_text_normalizer import VietnameseTextNormalizer
+from app.core.translation_validator import TranslationValidator
+from app.core.translation_quality_guard import TranslationQualityGuard
+from app.core.translation_post_editor import post_edit_translation
+from app.core.google_translate_fallback import GoogleTranslateFallback
+from app.core.subtitle_compactor import compact_blocks_for_tts
+from app.core.job_manager import JobManager
+from app.core.studio_sfx import generate_sfx_clip
+
+
+def has_chinese(text: str) -> bool:
+ """Returns True if string contains CJK Chinese characters."""
+ return bool(re.search(r"[\u4e00-\u9fff]", text))
+
+
+class CloudPipeline:
+ def __init__(
+ self,
+ base_dir: Optional[Path] = None,
+ log_callback: Optional[Callable[[str], None]] = None,
+ progress_callback: Optional[Callable[[int, str], None]] = None,
+ ffmpeg_path: str = "ffmpeg"
+ ):
+ self.base_dir = Path(base_dir or Path(__file__).resolve().parents[2])
+ self.output_dir = self.base_dir / "output"
+ self.temp_dir = self.base_dir / "temp"
+ self.output_dir.mkdir(parents=True, exist_ok=True)
+ self.temp_dir.mkdir(parents=True, exist_ok=True)
+
+ self.log_fn = log_callback or print
+ self.progress_fn = progress_callback or (lambda pct, stage: None)
+ self.ffmpeg_path = ffmpeg_path
+
+ self.asr_engine = CloudASREngine(log_fn=self._log)
+ self.ocr_engine = CloudOCREngine(log_fn=self._log)
+ self.tts_engine = CloudTTSEngine(log_fn=self._log, ffmpeg_path=self.ffmpeg_path)
+ self.normalizer = VietnameseTextNormalizer()
+ self.job_manager = JobManager.instance()
+ self.glossary = self._load_glossary()
+
+ def _log(self, msg: str):
+ self.log_fn(f"[Cloud Pipeline] {msg}")
+
+ def _load_glossary(self) -> Dict[str, str]:
+ glossary_path = self.base_dir / "glossary.json"
+ if glossary_path.exists():
+ try:
+ data = json.loads(glossary_path.read_text(encoding="utf-8"))
+ return data.get("glossary", {})
+ except Exception as e:
+ self._log(f"⚠️ Lỗi đọc glossary.json: {e}")
+ return {}
+
+ def calculate_region(
+ self,
+ video_path: Path,
+ region_preset: str = "custom",
+ custom_y_pct: float = 75.0,
+ custom_h_pct: float = 20.0,
+ custom_x_pct: float = 0.0,
+ custom_w_pct: float = 100.0
+ ) -> Tuple[int, int, int, int]:
+ vw, vh = 1920, 1080
+ try:
+ cap = cv2.VideoCapture(str(video_path))
+ w = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH))
+ h = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT))
+ cap.release()
+ if w > 0 and h > 0:
+ vw, vh = w, h
+ except Exception:
+ pass
+
+ if region_preset == "bottom_25":
+ x = 0
+ w = vw
+ y = int(vh * 0.72)
+ h = int(vh * 0.24)
+ elif region_preset == "bottom_15":
+ x = 0
+ w = vw
+ y = int(vh * 0.82)
+ h = int(vh * 0.15)
+ elif region_preset == "middle_25":
+ x = 0
+ w = vw
+ y = int(vh * 0.38)
+ h = int(vh * 0.24)
+ elif region_preset == "none":
+ return (0, 0, 0, 0)
+ elif region_preset == "custom":
+ x = int(vw * (custom_x_pct / 100.0))
+ w = int(vw * (custom_w_pct / 100.0))
+ y = int(vh * (custom_y_pct / 100.0))
+ h = int(vh * (custom_h_pct / 100.0))
+ else:
+ return (0, 0, 0, 0)
+
+ x = max(0, min(vw - 2, x))
+ y = max(0, min(vh - 2, y))
+ w = max(2, min(vw - x, w))
+ h = max(2, min(vh - y, h))
+ return (x, y, w, h)
+
+ def generate_preview_frame(
+ self,
+ video_path: str,
+ region_preset: str = "custom",
+ custom_y_pct: float = 75.0,
+ custom_h_pct: float = 20.0,
+ custom_x_pct: float = 0.0,
+ custom_w_pct: float = 100.0,
+ sec: float = 2.0,
+ return_type: str = "rgb",
+ # ── Logo overlay params ──
+ logo_enabled: bool = False,
+ logo_path: str = "",
+ logo_preset: str = "bottom_right",
+ logo_scale: float = 15.0,
+ logo_opacity: float = 1.0,
+ logo_x_pct: float = 80.0,
+ logo_y_pct: float = 80.0,
+ logo_chromakey: bool = True,
+ # ── CTA video overlay params (appears every 5s) ──
+ cta_enabled: bool = False,
+ cta_path: str = "",
+ cta_preset: str = "bottom_center",
+ cta_scale: float = 35.0,
+ cta_opacity: float = 1.0,
+ cta_x_pct: float = 50.0,
+ cta_y_pct: float = 85.0,
+ cta_interval: float = 5.0,
+ cta_duration: float = 2.0,
+ cta_chromakey: bool = True
+ ):
+ """Extracts a frame at timestamp sec and overlays bounding box + logo + CTA preview."""
+ v_file = Path(video_path)
+ if not v_file.exists():
+ return None
+
+ raw_frame_path = self.temp_dir / f"raw_frame_{int(time.time()*1000)}.jpg"
+ cmd = [
+ str(self.ffmpeg_path), "-ss", str(sec), "-y",
+ "-i", str(v_file),
+ "-frames:v", "1", "-q:v", "2",
+ str(raw_frame_path)
+ ]
+ try:
+ subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True)
+ frame = cv2.imread(str(raw_frame_path))
+ if raw_frame_path.exists():
+ raw_frame_path.unlink()
+ except Exception:
+ frame = None
+
+ if frame is None:
+ cap = cv2.VideoCapture(str(v_file))
+ fps = cap.get(cv2.CAP_PROP_FPS) or 25.0
+ target_frame = max(0, int(sec * fps))
+ cap.set(cv2.CAP_PROP_POS_FRAMES, target_frame)
+ ret, frame = cap.read()
+ cap.release()
+
+ if frame is None:
+ return None
+
+ x, y, w, h = self.calculate_region(
+ v_file, region_preset, custom_y_pct, custom_h_pct, custom_x_pct, custom_w_pct
+ )
+
+ if w > 0 and h > 0:
+ overlay = frame.copy()
+ cv2.rectangle(overlay, (x, y), (x + w, y + h), (0, 0, 255), -1)
+ cv2.addWeighted(overlay, 0.35, frame, 0.65, 0, frame)
+ cv2.rectangle(frame, (x, y), (x + w, y + h), (0, 255, 255), 3)
+
+ label = "VUNG CHE SUB CU & QUET OCR"
+ font = cv2.FONT_HERSHEY_SIMPLEX
+ font_scale = max(0.5, frame.shape[1] / 1200.0)
+ thickness = 2
+ (tw, th), _ = cv2.getTextSize(label, font, font_scale, thickness)
+
+ label_y = max(th + 10, y - 8)
+ cv2.rectangle(frame, (x, label_y - th - 6), (x + tw + 10, label_y + 4), (0, 0, 0), -1)
+ cv2.putText(frame, label, (x + 5, label_y), font, font_scale, (0, 255, 255), thickness, cv2.LINE_AA)
+
+ # ── Logo overlay preview ──
+ if logo_enabled and logo_path and Path(logo_path).exists():
+ try:
+ from app.core.logo_overlay import load_logo_rgba, overlay_logo_on_frame
+ _logo_rgba = load_logo_rgba(logo_path, chromakey=logo_chromakey)
+ if _logo_rgba is not None:
+ frame = overlay_logo_on_frame(
+ frame_bgr=frame,
+ logo_rgba=_logo_rgba,
+ preset=logo_preset,
+ scale_pct=float(logo_scale),
+ opacity=float(logo_opacity),
+ custom_x_pct=float(logo_x_pct),
+ custom_y_pct=float(logo_y_pct),
+ margin_pct=2.0
+ )
+ cv2.putText(frame, f"LOGO:{logo_preset} {logo_scale:.0f}%", (10, 30), cv2.FONT_HERSHEY_SIMPLEX, 0.7, (0, 255, 255), 2, cv2.LINE_AA)
+ except Exception as _e:
+ self._log(f"⚠️ Logo preview error: {_e}")
+
+ # ── CTA video overlay preview (every 5s) ──
+ if cta_enabled and cta_path and Path(cta_path).exists():
+ try:
+ from app.core.logo_overlay import load_cta_frame_rgba, overlay_cta_on_frame, should_show_cta_at_time
+ if should_show_cta_at_time(float(sec), float(cta_interval), float(cta_duration)):
+ # CTA time loops within CTA video duration
+ _cta_time = float(sec) % float(cta_interval)
+ # If CTA video is shorter than interval, loop inside
+ _cta_rgba = load_cta_frame_rgba(cta_path, cta_time_sec=_cta_time, chromakey=cta_chromakey)
+ if _cta_rgba is not None:
+ frame = overlay_cta_on_frame(
+ frame_bgr=frame,
+ cta_rgba=_cta_rgba,
+ preset=cta_preset,
+ scale_pct=float(cta_scale),
+ opacity=float(cta_opacity),
+ custom_x_pct=float(cta_x_pct),
+ custom_y_pct=float(cta_y_pct),
+ margin_pct=2.0
+ )
+ cv2.putText(frame, f"CTA:{cta_preset} {cta_scale:.0f}% every {cta_interval:.0f}s", (10, 60), cv2.FONT_HERSHEY_SIMPLEX, 0.6, (0, 255, 0), 2, cv2.LINE_AA)
+ else:
+ cv2.putText(frame, f"CTA: hidden at {sec:.1f}s (interval {cta_interval:.0f}s)", (10, 60), cv2.FONT_HERSHEY_SIMPLEX, 0.5, (0, 165, 255), 1, cv2.LINE_AA)
+ except Exception as _e:
+ self._log(f"⚠️ CTA preview error: {_e}")
+
+ if return_type == "base64":
+ _, buffer = cv2.imencode('.jpg', frame, [cv2.IMWRITE_JPEG_QUALITY, 85])
+ return base64.b64encode(buffer).decode('utf-8')
+
+ return cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)
+
+ def extract_subtitles_only(
+ self,
+ video_path: str,
+ mode: str = "asr",
+ source_lang: str = "zh",
+ region_preset: str = "custom",
+ custom_y_pct: float = 75.0,
+ custom_h_pct: float = 20.0,
+ custom_x_pct: float = 0.0,
+ custom_w_pct: float = 100.0
+ ) -> Optional[Dict[str, Any]]:
+ """
+ Runs Stage 0 -> Stage A -> Stage C.
+ Returns parsed subtitle blocks (original & translated) for Web Subtitle Editor.
+ """
+ v_path = Path(video_path)
+ if not v_path.exists():
+ self._log(f"❌ Video not found: {video_path}")
+ return None
+
+ video_stem = v_path.stem
+ vtd = self.temp_dir / video_stem
+ vtd.mkdir(parents=True, exist_ok=True)
+
+ self.progress_fn(5, "STAGE_0_PREPARE")
+ calc_region = self.calculate_region(
+ v_path, region_preset, custom_y_pct, custom_h_pct, custom_x_pct, custom_w_pct
+ )
+ x, y, w, h = calc_region
+
+ # 1. Extract audio
+ extracted_audio = vtd / "extracted_audio.wav"
+ self._log("⚡ Trích xuất âm thanh từ video gốc...")
+ cmd = [
+ str(self.ffmpeg_path), "-y", "-i", str(v_path),
+ "-vn", "-acodec", "pcm_s16le", "-ar", "16000", "-ac", "1",
+ str(extracted_audio)
+ ]
+ subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True)
+
+ # 2. OCR / ASR
+ self.progress_fn(25, "STAGE_A_OCR_ASR")
+ original_srt = vtd / "original.srt"
+ if mode == "ocr":
+ self._log("👁️ Quét chữ phụ đề bằng AI Vision Models...")
+ ocr_box = (x, y, w, h) if (w > 0 and h > 0) else None
+ ok = self.ocr_engine.scan_video_subtitles_to_srt(str(v_path), str(original_srt), blur_region=ocr_box)
+ if not ok or not original_srt.exists() or original_srt.stat().st_size < 10:
+ self._log("⚠️ OCR không phát hiện chữ -> Tự động chuyển sang Cloud ASR...")
+ ok = self.asr_engine.transcribe_audio_to_srt(str(extracted_audio), str(original_srt), source_lang=source_lang)
+ else:
+ self._log("🎙️ Nhận diện giọng nói bằng Cloud ASR (Groq Whisper Large-v3)...")
+ ok = self.asr_engine.transcribe_audio_to_srt(str(extracted_audio), str(original_srt), source_lang=source_lang)
+
+ if not ok or not original_srt.exists() or original_srt.stat().st_size < 10:
+ raise RuntimeError("Không thể trích xuất phụ đề từ video.")
+
+ # 3. Translation
+ self.progress_fn(50, "STAGE_C_TRANSLATION")
+ self._log("🌐 Dịch thuật bằng AI Translation Matrix...")
+ translated_srt = vtd / "translated.srt"
+ self._translate_srt_cloud(original_srt, translated_srt, source_lang=source_lang)
+
+ orig_blocks = self._parse_srt_blocks(original_srt.read_text(encoding="utf-8", errors="ignore"))
+ trans_blocks = self._parse_srt_blocks(translated_srt.read_text(encoding="utf-8", errors="ignore"))
+
+ combined = []
+ for ob in orig_blocks:
+ tb = next((t for t in trans_blocks if t["id"] == ob["id"]), None)
+ combined.append({
+ "id": ob["id"],
+ "timing": ob["timing"],
+ "start_ms": ob["start_ms"],
+ "end_ms": ob["end_ms"],
+ "original_text": ob["text"],
+ "vietnamese_text": tb["text"] if tb else ob["text"],
+ "sfx": ""
+ })
+
+ return {
+ "video_stem": video_stem,
+ "blocks": combined,
+ "original_srt": str(original_srt),
+ "translated_srt": str(translated_srt)
+ }
+
+ def render_from_subtitles(
+ self,
+ video_path: str,
+ subtitles_data: List[Dict[str, Any]],
+ voice: str = "vi-VN-NamMinhNeural",
+ speed: float = 1.0,
+ pitch: int = 0,
+ volume: int = 100,
+ region_preset: str = "custom",
+ sub_mask_mode: str = "box",
+ sub_style: str = "motion_drip",
+ custom_y_pct: float = 75.0,
+ custom_h_pct: float = 20.0,
+ custom_x_pct: float = 0.0,
+ custom_w_pct: float = 100.0,
+ ducking_ratio: float = 0.18,
+ enable_sfx: bool = True,
+ mute_original_audio: bool = False,
+ # ── Logo overlay ──
+ logo_enabled: bool = False,
+ logo_path: str = "",
+ logo_preset: str = "bottom_right",
+ logo_scale: float = 15.0,
+ logo_opacity: float = 1.0,
+ logo_x_pct: float = 80.0,
+ logo_y_pct: float = 80.0,
+ logo_chromakey: bool = True,
+ # ── CTA video overlay (every 5s) ──
+ cta_enabled: bool = False,
+ cta_path: str = "",
+ cta_preset: str = "bottom_center",
+ cta_scale: float = 35.0,
+ cta_opacity: float = 1.0,
+ cta_x_pct: float = 50.0,
+ cta_y_pct: float = 85.0,
+ cta_interval: float = 5.0,
+ cta_duration: float = 2.0,
+ cta_chromakey: bool = True
+ ) -> Optional[str]:
+ """Completes TTS synthesis, Ducking audio mix, and video rendering from user-reviewed subtitles."""
+ v_path = Path(video_path)
+ if not v_path.exists():
+ return None
+
+ video_stem = v_path.stem
+ vtd = self.temp_dir / video_stem
+ vtd.mkdir(parents=True, exist_ok=True)
+
+ start_time = time.time()
+ calc_region = self.calculate_region(
+ v_path, region_preset, custom_y_pct, custom_h_pct, custom_x_pct, custom_w_pct
+ )
+ x, y, w, h = calc_region
+
+ # Write edited SRT
+ translated_srt = vtd / "translated.srt"
+ srt_lines = []
+ for item in subtitles_data:
+ srt_lines.append(str(item["id"]))
+ srt_lines.append(item["timing"])
+ vi_text = self.normalizer.normalize(item["vietnamese_text"]) if hasattr(self.normalizer, "normalize") else item["vietnamese_text"]
+ srt_lines.append(vi_text)
+ srt_lines.append("")
+ translated_srt.write_text("\n".join(srt_lines), encoding="utf-8")
+
+ # 4. Cloud TTS
+ self.progress_fn(65, "STAGE_TTS_DUBBING")
+ self._log(f"🎙️ Tạo giọng đọc lồng tiếng ({voice}, speed={speed}x, volume={volume}%)...")
+ dubbing_wav = vtd / "dubbing.wav"
+ ok_tts = self.tts_engine.synthesize_srt_to_audio(
+ str(translated_srt),
+ str(dubbing_wav),
+ voice=voice,
+ speed=speed,
+ pitch=pitch,
+ volume=volume,
+ temp_dir=str(vtd / "tts_segments")
+ )
+ if not ok_tts or not dubbing_wav.exists():
+ raise RuntimeError("Lỗi tạo giọng đọc TTS.")
+
+ # 5. SFX & Audio Ducking
+ self.progress_fn(80, "STAGE_D_AUDIO_MIX")
+ extracted_audio = vtd / "extracted_audio.wav"
+ if not extracted_audio.exists():
+ cmd = [
+ str(self.ffmpeg_path), "-y", "-i", str(v_path),
+ "-vn", "-acodec", "pcm_s16le", "-ar", "16000", "-ac", "1",
+ str(extracted_audio)
+ ]
+ subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True)
+
+ mixed_audio = vtd / "mixed_final.wav"
+ if mute_original_audio:
+ self._log("🔇 Tắt hoàn toàn tiếng gốc — chỉ giữ giọng Việt (mute_original_audio=ON)")
+ # Dubbing wav đã được time-stretch canvas đúng duration, chỉ cần chuẩn hóa sample-rate
+ cmd_mix = [
+ str(self.ffmpeg_path), "-y",
+ "-i", str(dubbing_wav),
+ "-ac", "2", "-ar", "48000",
+ str(mixed_audio)
+ ]
+ subprocess.run(cmd_mix, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True)
+ else:
+ self._log("🎚️ Trộn nhạc nền (Auto Ducking) + Giọng đọc thuyết minh...")
+ filter_complex = (
+ f"[0:a]volume={ducking_ratio}[bg];"
+ f"[1:a]volume=1.0[dub];"
+ f"[bg][dub]amix=inputs=2:duration=longest:dropout_transition=2:normalize=0[aout]"
+ )
+ cmd_mix = [
+ str(self.ffmpeg_path), "-y",
+ "-i", str(extracted_audio),
+ "-i", str(dubbing_wav),
+ "-filter_complex", filter_complex,
+ "-map", "[aout]",
+ "-ac", "2", "-ar", "48000",
+ str(mixed_audio)
+ ]
+ subprocess.run(cmd_mix, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True)
+
+ # 6. Render Video
+ self.progress_fn(90, "STAGE_D_RENDER")
+ self._log("🎬 Render video hoàn thiện: Xóa sub cũ & Đè sub tiếng Việt...")
+ final_output = self.output_dir / f"studio_final_{v_path.name}"
+
+ # Probe dimensions with cv2 then ffprobe fallback (handles vertical & cv2-missing envs)
+ vw, vh = 1920, 1080
+ try:
+ cap = cv2.VideoCapture(str(v_path))
+ w_tmp = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH)) or 0
+ h_tmp = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT)) or 0
+ cap.release()
+ if w_tmp > 0 and h_tmp > 0:
+ vw, vh = w_tmp, h_tmp
+ else:
+ raise ValueError("cv2 returned 0")
+ except Exception:
+ try:
+ import shutil as _sh
+ _ffprobe = _sh.which("ffprobe") or str(Path(self.ffmpeg_path).parent / "ffprobe.exe") if Path(self.ffmpeg_path).exists() else "ffprobe"
+ if not Path(_ffprobe).exists():
+ _ffprobe = "ffprobe"
+ _res = subprocess.run([_ffprobe, "-v", "error", "-select_streams", "v:0", "-show_entries", "stream=width,height", "-of", "json", str(v_path)], capture_output=True, text=True, timeout=5)
+ _j = json.loads(_res.stdout or "{}")
+ _ws = _j.get("streams", [{}])[0]
+ if _ws.get("width") and _ws.get("height"):
+ vw, vh = int(_ws["width"]), int(_ws["height"])
+ except Exception:
+ pass
+
+ translated_ass = vtd / "translated.ass"
+ self._convert_srt_to_ass(translated_srt, translated_ass, vw, vh, x, y, w, h, sub_style)
+
+ ass_escaped = str(translated_ass).replace("\\", "/").replace(":", "\\:")
+ sub_filter = f"subtitles='{ass_escaped}'"
+
+ vf_filters = []
+ if w > 0 and h > 0:
+ if sub_mask_mode == "delogo":
+ safe_x = max(2, x)
+ safe_y = max(2, y)
+ safe_w = max(4, min(w, vw - safe_x - 2))
+ safe_h = max(4, min(h, vh - safe_y - 2))
+ vf_filters.append(f"delogo=x={safe_x}:y={safe_y}:w={safe_w}:h={safe_h}")
+ elif sub_mask_mode == "box":
+ vf_filters.append(f"drawbox=x={x}:y={y}:w={w}:h={h}:color=black@0.92:t=fill")
+
+ # Determine logo & CTA overlay enabled (with fallback to bundled assets)
+ _logo_enabled = bool(logo_enabled and logo_path and Path(logo_path).exists())
+ if logo_enabled and not _logo_enabled:
+ _fb = self.base_dir / "assets" / "logo_ins_drippy4.png"
+ if _fb.exists():
+ _logo_enabled = True
+ logo_path = str(_fb)
+ _logo_path = Path(logo_path) if _logo_enabled else None
+ _cta_enabled = bool(cta_enabled and cta_path and Path(cta_path).exists())
+ if cta_enabled and not _cta_enabled:
+ _fb2 = self.base_dir / "assets" / "cta_ins_drippy4.mp4"
+ if _fb2.exists():
+ _cta_enabled = True
+ cta_path = str(_fb2)
+ _cta_path = Path(cta_path) if _cta_enabled else None
+
+ # If any overlay enabled, we need filter_complex
+ if _logo_enabled or _cta_enabled:
+ if _logo_enabled:
+ self._log(f"🖼️ Overlay logo: {Path(logo_path).name} preset={logo_preset} scale={logo_scale}% opacity={logo_opacity} chromakey={logo_chromakey}")
+ if _cta_enabled:
+ self._log(f"🎬 Overlay CTA: {Path(cta_path).name} preset={cta_preset} scale={cta_scale}% every {cta_interval}s for {cta_duration}s chromakey={cta_chromakey}")
+ # Build filter parts step-by-step. Inputs: 0:video, 1:audio, 2:logo(if), 3:cta(if)
+ # Determine indices
+ _logo_idx = 2 if _logo_enabled else None
+ _cta_idx = None
+ if _cta_enabled:
+ _cta_idx = 3 if _logo_enabled else 2
+ # Video preprocessing (delogo/box) -> [base]
+ _filter_parts = []
+ if vf_filters and len([f for f in vf_filters if f != sub_filter]) > 0:
+ _pre_vf = ",".join([f for f in vf_filters if f != sub_filter])
+ _filter_parts.append(f"[0:v]{_pre_vf}[base]")
+ _cur = "base"
+ else:
+ _filter_parts.append("[0:v]null[base]")
+ _cur = "base"
+ # Logo overlay
+ if _logo_enabled:
+ _logo_w_orig, _logo_h_orig = 2400, 1792
+ try:
+ _t = cv2.imread(str(_logo_path), cv2.IMREAD_UNCHANGED)
+ if _t is not None:
+ _logo_h_orig, _logo_w_orig = _t.shape[:2]
+ except Exception:
+ pass
+ _target_w = vw * float(logo_scale) / 100.0
+ _sf = max(0.02, min(0.5, _target_w / float(max(1, _logo_w_orig))))
+ _logo_vf_parts = []
+ if logo_chromakey:
+ _logo_vf_parts.append("colorkey=0x00FF00:0.3:0.1")
+ _logo_vf_parts.append("format=rgba")
+ _logo_vf_parts.append(f"scale=iw*{_sf:.4f}:ih*{_sf:.4f}:flags=lanczos")
+ if float(logo_opacity) < 0.99:
+ _logo_vf_parts.append(f"colorchannelmixer=aa={float(logo_opacity):.2f}")
+ _logo_vf = ",".join(_logo_vf_parts)
+ _m = 2.0
+ if logo_preset == "top_left":
+ _lx, _ly = f"W*{_m/100:.3f}", f"H*{_m/100:.3f}"
+ elif logo_preset == "top_right":
+ _lx, _ly = f"W-w-W*{_m/100:.3f}", f"H*{_m/100:.3f}"
+ elif logo_preset == "bottom_left":
+ _lx, _ly = f"W*{_m/100:.3f}", f"H-h-H*{_m/100:.3f}"
+ elif logo_preset == "bottom_right":
+ _lx, _ly = f"W-w-W*{_m/100:.3f}", f"H-h-H*{_m/100:.3f}"
+ elif logo_preset == "center":
+ _lx, _ly = "(W-w)/2", "(H-h)/2"
+ elif logo_preset == "top_center":
+ _lx, _ly = "(W-w)/2", f"H*{_m/100:.3f}"
+ elif logo_preset == "bottom_center":
+ _lx, _ly = "(W-w)/2", f"H-h-H*{_m/100:.3f}"
+ elif logo_preset == "custom":
+ _lx, _ly = f"W*{float(logo_x_pct)/100:.4f}", f"H*{float(logo_y_pct)/100:.4f}"
+ else:
+ _lx, _ly = f"W-w-W*{_m/100:.3f}", f"H-h-H*{_m/100:.3f}"
+ _filter_parts.append(f"[{_logo_idx}:v]{_logo_vf}[logo]")
+ _next = "with_logo" if _cta_enabled or sub_filter else "v"
+ _filter_parts.append(f"[{_cur}][logo]overlay={_lx}:{_ly}:format=rgb[{_next}]")
+ _cur = _next
+ # CTA overlay (periodic every interval)
+ if _cta_enabled:
+ # Probe CTA size
+ _cta_w_orig, _cta_h_orig = 1080, 1920
+ try:
+ _cap = cv2.VideoCapture(str(_cta_path))
+ _cta_w_orig = int(_cap.get(cv2.CAP_PROP_FRAME_WIDTH)) or _cta_w_orig
+ _cta_h_orig = int(_cap.get(cv2.CAP_PROP_FRAME_HEIGHT)) or _cta_h_orig
+ _cap.release()
+ except Exception:
+ pass
+ # TikTok CTA bump: enforce minimum 42% width for mobile visibility
+ effective_cta_scale = max(float(cta_scale), 42.0)
+ _cta_target_w = vw * effective_cta_scale / 100.0
+ _cta_sf = max(0.05, min(1.0, _cta_target_w / float(max(1, _cta_w_orig))))
+ # Clamp height to 80% of video height to avoid overflow (CTA vertical on horizontal video)
+ _cta_target_h = _cta_h_orig * _cta_sf
+ _max_h = vh * 0.90
+ if _cta_target_h > _max_h:
+ _cta_sf = _max_h / float(max(1, _cta_h_orig))
+ _cta_target_w = _cta_w_orig * _cta_sf
+ _cta_vf_parts = []
+ if cta_chromakey:
+ _cta_vf_parts.append("colorkey=0x00FF00:0.3:0.1")
+ _cta_vf_parts.append("format=rgba")
+ _cta_vf_parts.append(f"scale=iw*{_cta_sf:.4f}:ih*{_cta_sf:.4f}:flags=lanczos")
+ if float(cta_opacity) < 0.99:
+ _cta_vf_parts.append(f"colorchannelmixer=aa={float(cta_opacity):.2f}")
+ _cta_vf = ",".join(_cta_vf_parts)
+ _m2 = 2.0
+ if cta_preset == "top_left":
+ _cx, _cy = f"W*{_m2/100:.3f}", f"H*{_m2/100:.3f}"
+ elif cta_preset == "top_right":
+ _cx, _cy = f"W-w-W*{_m2/100:.3f}", f"H*{_m2/100:.3f}"
+ elif cta_preset == "bottom_left":
+ _cx, _cy = f"W*{_m2/100:.3f}", f"H-h-H*{_m2/100:.3f}"
+ elif cta_preset == "bottom_right":
+ _cx, _cy = f"W-w-W*{_m2/100:.3f}", f"H-h-H*{_m2/100:.3f}"
+ elif cta_preset == "center":
+ _cx, _cy = "(W-w)/2", "(H-h)/2"
+ elif cta_preset == "top_center":
+ _cx, _cy = "(W-w)/2", f"H*{_m2/100:.3f}"
+ elif cta_preset == "bottom_center":
+ _cx, _cy = "(W-w)/2", f"H-h-H*{_m2/100:.3f}"
+ elif cta_preset == "custom":
+ _cx, _cy = f"W*{float(cta_x_pct)/100:.4f}", f"H*{float(cta_y_pct)/100:.4f}"
+ else:
+ _cx, _cy = "(W-w)/2", f"H-h-H*{_m2/100:.3f}"
+ _enable = f"lt(mod(t\\,{float(cta_interval)})\\,{float(cta_duration)})"
+ _filter_parts.append(f"[{_cta_idx}:v]{_cta_vf}[cta]")
+ _next2 = "v" if not sub_filter else "with_cta"
+ # Use escaped comma for FFmpeg enable expression
+ _filter_parts.append(f"[{_cur}][cta]overlay={_cx}:{_cy}:format=rgb:enable='{_enable}'[{_next2}]")
+ _cur = _next2
+ # TikTok polish: CFR 30 + even dims + light sharpen BEFORE subtitles (keeps text razor sharp)
+ tiktok_polish = "fps=30:round=near,scale=trunc(iw/2)*2:trunc(ih/2)*2:flags=lanczos+accurate_rnd+full_chroma_int:sws_dither=ed,unsharp=3:3:0.35:3:3:0.0"
+ if sub_filter:
+ _filter_parts.append(f"[{_cur}]{tiktok_polish}[polished]")
+ _cur = "polished"
+ _filter_parts.append(f"[{_cur}]{sub_filter}[v]")
+ _cur = "v"
+ else:
+ _filter_parts.append(f"[{_cur}]{tiktok_polish}[v]")
+ _cur = "v"
+ _filter_complex = ";".join(_filter_parts)
+ # Build ffmpeg inputs: video, audio, logo(if), cta(if) - add -shortest when looped overlays present
+ cmd_inputs = [str(self.ffmpeg_path), "-y", "-fflags", "+genpts", "-avoid_negative_ts", "make_zero", "-i", str(v_path), "-i", str(mixed_audio)]
+ if _logo_enabled:
+ cmd_inputs += ["-loop", "1", "-i", str(_logo_path)]
+ if _cta_enabled:
+ cmd_inputs += ["-stream_loop", "999", "-i", str(_cta_path)]
+ _extra = ["-shortest"] if (_logo_enabled or _cta_enabled) else []
+ # TikTok spec: High Profile yuv420p, 30 CFR, 48k AAC, faststart, accurate sync
+ tiktok_video_args = ["-r", "30", "-c:v", "libx264", "-preset", "medium", "-crf", "18", "-profile:v", "high", "-level", "4.1", "-pix_fmt", "yuv420p", "-g", "60", "-keyint_min", "30", "-sc_threshold", "0", "-x264-params", "ref=4:bframes=2:me=hex:subme=7:psy=1:psy-rd=0.8:aq-mode=2", "-colorspace", "bt709", "-color_primaries", "bt709", "-color_trc", "bt709", "-color_range", "tv"]
+ tiktok_audio_args = ["-c:a", "aac", "-profile:a", "aac_low", "-ar", "48000", "-ac", "2", "-b:a", "192k", "-af", "aresample=async=1:min_hard_comp=0.100000:first_pts=0"]
+ tiktok_mux_args = ["-movflags", "+faststart", "-fflags", "+genpts", "-max_interleave_delta", "100M", "-vsync", "cfr", "-fps_mode", "cfr"]
+ cmd_render = cmd_inputs + ["-filter_complex", _filter_complex, "-map", "[v]", "-map", "1:a:0"] + _extra + tiktok_video_args + tiktok_audio_args + tiktok_mux_args + ["-shortest", str(final_output)]
+ res = subprocess.run(cmd_render, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True)
+ if res.returncode != 0:
+ self._log(f"⚠️ Overlay render failed ({res.stderr[:300]}), fallback to TikTok-spec normal render...")
+ # Rebuild VF with TikTok polish inserted before subs (same as non-overlay branch)
+ tiktok_polish_fb = "fps=30:round=near,scale=trunc(iw/2)*2:trunc(ih/2)*2:flags=lanczos+accurate_rnd+full_chroma_int:sws_dither=ed,unsharp=3:3:0.35:3:3:0.0"
+ vf_without_sub_fb = [f for f in vf_filters if f != sub_filter]
+ ordered_fb = []
+ if vf_without_sub_fb:
+ ordered_fb.extend(vf_without_sub_fb)
+ ordered_fb.append(tiktok_polish_fb)
+ ordered_fb.append(sub_filter)
+ final_vf = ",".join(ordered_fb)
+ cmd_render = [
+ str(self.ffmpeg_path), "-y",
+ "-fflags", "+genpts", "-avoid_negative_ts", "make_zero",
+ "-i", str(v_path),
+ "-i", str(mixed_audio),
+ "-vf", final_vf,
+ "-map", "0:v:0", "-map", "1:a:0",
+ "-r", "30",
+ "-c:v", "libx264", "-preset", "medium", "-crf", "18",
+ "-profile:v", "high", "-level", "4.1", "-pix_fmt", "yuv420p",
+ "-g", "60", "-keyint_min", "30", "-sc_threshold", "0",
+ "-x264-params", "ref=4:bframes=2:me=hex:subme=7:psy=1:psy-rd=0.8:aq-mode=2",
+ "-colorspace", "bt709", "-color_primaries", "bt709", "-color_trc", "bt709", "-color_range", "tv",
+ "-c:a", "aac", "-profile:a", "aac_low", "-ar", "48000", "-ac", "2", "-b:a", "192k",
+ "-af", "aresample=async=1:min_hard_comp=0.100000:first_pts=0",
+ "-movflags", "+faststart",
+ "-fflags", "+genpts", "-max_interleave_delta", "100M",
+ "-vsync", "cfr", "-fps_mode", "cfr",
+ "-shortest",
+ str(final_output)
+ ]
+ res = subprocess.run(cmd_render, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True)
+ if res.returncode != 0:
+ self._log(f"⚠️ Fallback render không mask: {res.stderr[:120]}")
+ fallback_cmd = [
+ str(self.ffmpeg_path), "-y",
+ "-fflags", "+genpts",
+ "-i", str(v_path),
+ "-i", str(mixed_audio),
+ "-vf", f"{tiktok_polish_fb},{sub_filter}",
+ "-map", "0:v:0", "-map", "1:a:0",
+ "-r", "30",
+ "-c:v", "libx264", "-preset", "medium", "-crf", "18",
+ "-profile:v", "high", "-pix_fmt", "yuv420p",
+ "-c:a", "aac", "-profile:a", "aac_low", "-ar", "48000", "-ac", "2", "-b:a", "192k",
+ "-movflags", "+faststart",
+ "-shortest",
+ str(final_output)
+ ]
+ subprocess.run(fallback_cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True)
+ else:
+ # TikTok polish inserted before subtitles for sharpness + CFR + even dims
+ tiktok_polish = "fps=30:round=near,scale=trunc(iw/2)*2:trunc(ih/2)*2:flags=lanczos+accurate_rnd+full_chroma_int:sws_dither=ed,unsharp=3:3:0.35:3:3:0.0"
+ # Order: mask (delogo/box) -> polish -> subtitles
+ vf_without_sub = [f for f in vf_filters if f != sub_filter]
+ # Rebuild vf in TikTok-optimal order
+ ordered_vf = []
+ if vf_without_sub:
+ ordered_vf.extend(vf_without_sub)
+ ordered_vf.append(tiktok_polish)
+ ordered_vf.append(sub_filter)
+ final_vf = ",".join(ordered_vf)
+
+ cmd_render = [
+ str(self.ffmpeg_path), "-y",
+ "-fflags", "+genpts", "-avoid_negative_ts", "make_zero",
+ "-i", str(v_path),
+ "-i", str(mixed_audio),
+ "-vf", final_vf,
+ "-map", "0:v:0", "-map", "1:a:0",
+ "-r", "30",
+ "-c:v", "libx264", "-preset", "medium", "-crf", "18",
+ "-profile:v", "high", "-level", "4.1", "-pix_fmt", "yuv420p",
+ "-g", "60", "-keyint_min", "30", "-sc_threshold", "0",
+ "-x264-params", "ref=4:bframes=2:me=hex:subme=7:psy=1:psy-rd=0.8:aq-mode=2",
+ "-colorspace", "bt709", "-color_primaries", "bt709", "-color_trc", "bt709", "-color_range", "tv",
+ "-c:a", "aac", "-profile:a", "aac_low", "-ar", "48000", "-ac", "2", "-b:a", "192k",
+ "-af", "aresample=async=1:min_hard_comp=0.100000:first_pts=0",
+ "-movflags", "+faststart",
+ "-fflags", "+genpts", "-max_interleave_delta", "100M",
+ "-vsync", "cfr", "-fps_mode", "cfr",
+ "-shortest",
+ str(final_output)
+ ]
+ res = subprocess.run(cmd_render, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True)
+ if res.returncode != 0:
+ self._log(f"⚠️ Fallback render không mask: {res.stderr[:120]} | retry with minimal TikTok spec")
+ # Minimal fallback: at least ensure TikTok audio/video spec
+ fallback_cmd = [
+ str(self.ffmpeg_path), "-y",
+ "-fflags", "+genpts",
+ "-i", str(v_path),
+ "-i", str(mixed_audio),
+ "-vf", f"{tiktok_polish},{sub_filter}",
+ "-map", "0:v:0", "-map", "1:a:0",
+ "-r", "30",
+ "-c:v", "libx264", "-preset", "medium", "-crf", "18",
+ "-profile:v", "high", "-pix_fmt", "yuv420p",
+ "-c:a", "aac", "-profile:a", "aac_low", "-ar", "48000", "-ac", "2", "-b:a", "192k",
+ "-movflags", "+faststart",
+ "-shortest",
+ str(final_output)
+ ]
+ subprocess.run(fallback_cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True)
+
+ total_elapsed = time.time() - start_time
+ self.progress_fn(100, "DONE")
+ self._log(f"🎉 HOÀN THÀNH XUẤT SẮC TRONG {total_elapsed:.1f}s!")
+ self._log(f"📁 Video đã lưu tại: {final_output}")
+ return str(final_output)
+
+ def run_video(
+ self,
+ video_path: str,
+ mode: str = "asr",
+ source_lang: str = "zh",
+ voice: str = "vi-VN-NamMinhNeural",
+ speed: float = 1.0,
+ pitch: int = 0,
+ volume: int = 100,
+ region_preset: str = "custom",
+ sub_mask_mode: str = "box",
+ sub_style: str = "motion_drip",
+ custom_y_pct: float = 75.0,
+ custom_h_pct: float = 20.0,
+ custom_x_pct: float = 0.0,
+ custom_w_pct: float = 100.0,
+ ducking_ratio: float = 0.18,
+ enable_sfx: bool = True,
+ mute_original_audio: bool = False,
+ logo_enabled: bool = False,
+ logo_path: str = "",
+ logo_preset: str = "bottom_right",
+ logo_scale: float = 15.0,
+ logo_opacity: float = 1.0,
+ logo_x_pct: float = 80.0,
+ logo_y_pct: float = 80.0,
+ logo_chromakey: bool = True,
+ cta_enabled: bool = False,
+ cta_path: str = "",
+ cta_preset: str = "bottom_center",
+ cta_scale: float = 35.0,
+ cta_opacity: float = 1.0,
+ cta_x_pct: float = 50.0,
+ cta_y_pct: float = 85.0,
+ cta_interval: float = 5.0,
+ cta_duration: float = 2.0,
+ cta_chromakey: bool = True
+ ) -> Optional[str]:
+ """Full 1-Click End-to-End Automatic Dubbing Pipeline."""
+ v_path = Path(video_path)
+ if not v_path.exists():
+ self._log(f"❌ Video not found: {video_path}")
+ return None
+
+ # Register into SQLite Job DB
+ try:
+ self.job_manager.register_job(str(v_path))
+ except Exception:
+ pass
+
+ try:
+ # 1. Extract subtitles
+ subs_res = self.extract_subtitles_only(
+ video_path=video_path,
+ mode=mode,
+ source_lang=source_lang,
+ region_preset=region_preset,
+ custom_y_pct=custom_y_pct,
+ custom_h_pct=custom_h_pct,
+ custom_x_pct=custom_x_pct,
+ custom_w_pct=custom_w_pct
+ )
+ if not subs_res:
+ return None
+
+ # 2. Render from subtitles
+ return self.render_from_subtitles(
+ video_path=video_path,
+ subtitles_data=subs_res["blocks"],
+ voice=voice,
+ speed=speed,
+ pitch=pitch,
+ volume=volume,
+ region_preset=region_preset,
+ sub_mask_mode=sub_mask_mode,
+ sub_style=sub_style,
+ custom_y_pct=custom_y_pct,
+ custom_h_pct=custom_h_pct,
+ custom_x_pct=custom_x_pct,
+ custom_w_pct=custom_w_pct,
+ ducking_ratio=ducking_ratio,
+ enable_sfx=enable_sfx,
+ mute_original_audio=mute_original_audio,
+ logo_enabled=logo_enabled,
+ logo_path=logo_path,
+ logo_preset=logo_preset,
+ logo_scale=logo_scale,
+ logo_opacity=logo_opacity,
+ logo_x_pct=logo_x_pct,
+ logo_y_pct=logo_y_pct,
+ logo_chromakey=logo_chromakey,
+ cta_enabled=cta_enabled,
+ cta_path=cta_path,
+ cta_preset=cta_preset,
+ cta_scale=cta_scale,
+ cta_opacity=cta_opacity,
+ cta_x_pct=cta_x_pct,
+ cta_y_pct=cta_y_pct,
+ cta_interval=cta_interval,
+ cta_duration=cta_duration,
+ cta_chromakey=cta_chromakey
+ )
+ except Exception as e:
+ self._log(f"❌ LỖI PIPELINE: {str(e)}")
+ self.progress_fn(0, "FAILED")
+ return None
+
+ def _convert_srt_to_ass(
+ self,
+ srt_path: Path,
+ ass_path: Path,
+ vw: int,
+ vh: int,
+ box_x: int,
+ box_y: int,
+ box_w: int,
+ box_h: int,
+ sub_style: str = "motion_drip"
+ ):
+ """Converts SRT to styled ASS format for TikTok/Shorts typography."""
+ content = srt_path.read_text(encoding="utf-8", errors="ignore")
+
+ # TikTok safe zone: keep subtitle inside 82% height, 7% bottom margin, larger font for mobile legibility
+ safe_margin_v = max(42, int(vh * 0.07))
+ raw_margin_v = int(vh - (box_y + box_h * 0.75)) if box_h > 0 else safe_margin_v
+ margin_v = max(safe_margin_v, raw_margin_v)
+ # Clamp to avoid bottom UI overlap
+ max_bottom = int(vh * 0.12)
+ if margin_v < max_bottom:
+ margin_v = max_bottom
+ # Larger font for TikTok: vertical 56-72, horizontal 48-64
+ if box_h > 0:
+ base = int(box_h * 0.52)
+ if vh > vw: # vertical TikTok
+ font_size = max(42, min(72, base))
+ else:
+ font_size = max(38, min(64, base))
+ else:
+ font_size = 58 if vh > vw else 50
+
+ # Typography Color & Outline palettes - TikTok optimized for high compression
+ if sub_style == "motion_drip" or sub_style == "yellow_bold":
+ font_color = "&H0000FFFF" # Bright Yellow
+ outline_color = "&H00000000" # Pure Black
+ back_color = "&H99000000"
+ outline = 5.0
+ shadow = 2.2
+ elif sub_style == "neon_cyan":
+ font_color = "&H00FFFF00" # Cyan
+ outline_color = "&H00000000"
+ back_color = "&H99000000"
+ outline = 5.0
+ shadow = 2.2
+ elif sub_style == "capsule_tag":
+ font_color = "&H00FFFFFF" # White
+ outline_color = "&H00111111"
+ back_color = "&HBB000000"
+ outline = 3.2
+ shadow = 0.0
+ else: # white_bold
+ font_color = "&H00FFFFFF" # Crisp White
+ outline_color = "&H00000000"
+ back_color = "&H99000000"
+ outline = 4.8
+ shadow = 2.0
+
+ ass_header = f"""[Script Info]
+ScriptType: v4.00+
+PlayResX: {vw}
+PlayResY: {vh}
+ScaledBorderAndShadow: yes
+
+[V4+ Styles]
+Format: Name, Fontname, Fontsize, PrimaryColour, SecondaryColour, OutlineColour, BackColour, Bold, Italic, Underline, StrikeOut, ScaleX, ScaleY, Spacing, Angle, BorderStyle, Outline, Shadow, Alignment, MarginL, MarginR, MarginV, Encoding
+Style: Default,DejaVu Sans,{font_size},{font_color},&H000000FF,{outline_color},{back_color},1,0,0,0,100,100,0,0,1,{outline},{shadow},2,40,40,{margin_v},1
+
+[Events]
+Format: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text
+"""
+
+ events = []
+ pattern = r"(\d+)\s+(\d{2}:\d{2}:\d{2}[.,]\d{3})\s*-->\s*(\d{2}:\d{2}:\d{2}[.,]\d{3})\s*\n(.*?)(?=\n\s*\d+\s+\d{2}:\d{2}:\d{2}[.,]\d{3}\s*-->|\Z)"
+ for m in re.finditer(pattern, content, re.DOTALL):
+ start_str = m.group(2).strip().replace(",", ".")
+ end_str = m.group(3).strip().replace(",", ".")
+
+ s_parts = start_str.split(":")
+ e_parts = end_str.split(":")
+ s_ass = f"{int(s_parts[0])}:{s_parts[1]}:{float(s_parts[2]):05.2f}"
+ e_ass = f"{int(e_parts[0])}:{e_parts[1]}:{float(e_parts[2]):05.2f}"
+
+ text = " ".join(line.strip() for line in m.group(4).splitlines() if line.strip())
+ if text:
+ events.append(f"Dialogue: 0,{s_ass},{e_ass},Default,,0,0,0,,{text}")
+
+ ass_path.write_text(ass_header + "\n".join(events), encoding="utf-8")
+
+ def _parse_srt_blocks(self, content: str) -> List[Dict]:
+ pattern = r"(\d+)\s+(\d{2}:\d{2}:\d{2}[.,]\d{3})\s*-->\s*(\d{2}:\d{2}:\d{2}[.,]\d{3})\s*\n(.*?)(?=\n\s*\d+\s+\d{2}:\d{2}:\d{2}[.,]\d{3}\s*-->|\Z)"
+ blocks = []
+ for m in re.finditer(pattern, content, re.DOTALL):
+ text = " ".join(line.strip() for line in m.group(4).splitlines() if line.strip())
+ start_str = m.group(2).strip()
+ end_str = m.group(3).strip()
+ if text:
+ blocks.append({
+ "id": int(m.group(1)),
+ "timing": f"{start_str} --> {end_str}",
+ "text": text,
+ "start_ms": self._ts_to_ms(start_str),
+ "end_ms": self._ts_to_ms(end_str),
+ })
+ return blocks
+
+ @staticmethod
+ def _ts_to_ms(ts: str) -> int:
+ ts = ts.strip().replace(".", ",")
+ m = re.match(r"(\d+):(\d+):(\d+)[,](\d+)", ts)
+ if m:
+ h, mins, s, ms = map(int, m.groups())
+ return ((h * 3600 + mins * 60 + s) * 1000) + ms
+ return 0
+
+ def _build_system_prompt(self) -> str:
+ glossary_sample = ", ".join([f"{k} -> {v}" for k, v in list(self.glossary.items())[:35]])
+ return (
+ "Bạn là chuyên gia dịch thuật video Sneaker, Thời trang Streetwear, Review sản phẩm từ tiếng Trung/Anh sang tiếng Việt tự nhiên, sành điệu, bắt trend Gen Z.\n"
+ "QUY TẮC BẮT BUỘC:\n"
+ "1. Dịch từng dòng theo cấu trúc: [N] Câu dịch tiếng Việt hoàn chỉnh.\n"
+ "2. Giữ nguyên thuật ngữ & thương hiệu tiếng Anh (Nike, Jordan, Yeezy, BAPE, Supreme, Rick Owens, outfit, fit, drip, full box, collab, signature...). \n"
+ f"3. Bắt buộc áp dụng từ điển chuyên ngành: {glossary_sample}\n"
+ "4. Dịch ĐẦY ĐỦ Ý NGHĨA trọn vẹn của câu, giữ đủ các từ khoá, KHÔNG bỏ lửng hay cắt cụt mất từ ở cuối câu.\n"
+ "5. KHÔNG giải thích, CHỈ trả về danh sách các dòng [N] Tiếng Việt."
+ )
+
+ def _translate_srt_cloud(self, srt_in: Path, srt_out: Path, source_lang: str = "zh"):
+ content = srt_in.read_text(encoding="utf-8", errors="ignore")
+ blocks = self._parse_srt_blocks(content)
+
+ if not blocks:
+ srt_out.write_text(content, encoding="utf-8")
+ return
+
+ system_prompt = self._build_system_prompt()
+ validator = TranslationValidator()
+ trans_map = {}
+
+ # Phase 1: Batch translation in chunks of 20
+ chunk_size = 20
+ for i in range(0, len(blocks), chunk_size):
+ chunk = blocks[i:i + chunk_size]
+ chunk_result = self._translate_chunk_with_retry(chunk, system_prompt, validator, max_retries=2)
+ trans_map.update(chunk_result)
+
+ # Phase 2: Retry missing / Chinese-leaked blocks
+ missing_blocks = [b for b in blocks if (b["id"] not in trans_map or has_chinese(trans_map.get(b["id"], "")))]
+ if missing_blocks:
+ self._log(f"🔄 Đang hoàn thiện nốt {len(missing_blocks)} câu dịch còn lại...")
+ retry_result = self._translate_chunk_with_retry(missing_blocks, system_prompt, validator, max_retries=2)
+ trans_map.update(retry_result)
+
+ # Phase 3: Google Translate fallback
+ still_missing = [b for b in blocks if (b["id"] not in trans_map or has_chinese(trans_map.get(b["id"], "")))]
+ if still_missing:
+ self._log(f"🌐 {len(still_missing)} câu cần fallback Google Translate...")
+ google_fb = GoogleTranslateFallback(log_fn=self._log)
+ google_result = google_fb.translate_blocks(still_missing)
+ for bid_str, text in google_result.items():
+ trans_map[int(bid_str)] = text
+
+ # Phase 4: Post-editing
+ self._log("✨ Chạy Post-Editor sửa lỗi dịch...")
+ for b in blocks:
+ bid = b["id"]
+ if bid in trans_map:
+ trans_map[bid] = post_edit_translation(b["text"], trans_map[bid])
+
+ # Phase 5: Quality Guard
+ self._log("🛡️ Quality Guard kiểm tra chất lượng bản dịch...")
+ guard = TranslationQualityGuard(min_score=70)
+ guard_dict = {str(b["id"]): trans_map.get(b["id"], b["text"]) for b in blocks}
+ fixed_dict, quality_report = guard.audit_and_fix(blocks, guard_dict)
+
+ for b in blocks:
+ bid_str = str(b["id"])
+ if bid_str in fixed_dict:
+ trans_map[b["id"]] = fixed_dict[bid_str]
+
+ # Phase 6: Timing compaction
+ self._log("⏱️ Rút gọn câu dịch cho vừa timeline TTS...")
+ compact_dict = {str(b["id"]): trans_map.get(b["id"], b["text"]) for b in blocks}
+ compact_dict = compact_blocks_for_tts(blocks, compact_dict)
+ for b in blocks:
+ bid_str = str(b["id"])
+ if bid_str in compact_dict:
+ trans_map[b["id"]] = compact_dict[bid_str]
+
+ # Phase 7: Normalize + Output SRT
+ out_lines = []
+ for b in blocks:
+ vi_text = trans_map.get(b["id"], b["text"])
+ vi_text = self.normalizer.normalize(vi_text) if hasattr(self.normalizer, "normalize") else vi_text
+ out_lines.append(str(b["id"]))
+ out_lines.append(b["timing"])
+ out_lines.append(vi_text)
+ out_lines.append("")
+
+ srt_out.write_text("\n".join(out_lines), encoding="utf-8")
+ self._log(f"✅ Dịch hoàn tất {len(blocks)} câu với 7 bước kiểm tra chất lượng.")
+
+ def _translate_chunk_with_retry(
+ self,
+ chunk: List[Dict],
+ system_prompt: str,
+ validator: TranslationValidator,
+ max_retries: int = 2
+ ) -> Dict[int, str]:
+ prompt_lines = [f"[{b['id']}] {b['text']}" for b in chunk]
+ full_transcript = "\n".join(prompt_lines)
+ result = {}
+
+ for attempt in range(max_retries + 1):
+ raw_res = self._direct_25_model_translate(system_prompt, full_transcript)
+ cleaned_res = re.sub(r".*?", "", raw_res, flags=re.DOTALL).strip()
+
+ attempt_map = {}
+ for line in cleaned_res.splitlines():
+ m = re.match(r"^\s*\[(\d+)\]\s*(.*)$", line.strip())
+ if m:
+ attempt_map[int(m.group(1))] = m.group(2).strip()
+
+ val_dict = {str(b["id"]): attempt_map.get(b["id"], "") for b in chunk if b["id"] in attempt_map}
+ val_chunk = [b for b in chunk if b["id"] in attempt_map]
+
+ if val_chunk and val_dict:
+ is_valid, error_msg = validator.validate(val_chunk, val_dict)
+ if is_valid:
+ result.update(attempt_map)
+ return result
+ else:
+ for bid, text in attempt_map.items():
+ src_block = next((b for b in chunk if b["id"] == bid), None)
+ if src_block:
+ ok, _ = validator.validate_single_block(bid, src_block["text"], text)
+ if ok:
+ result[bid] = text
+ else:
+ result.update(attempt_map)
+
+ return result
+
+ def _direct_25_model_translate(self, system_prompt: str, user_content: str) -> str:
+ messages = [
+ {"role": "system", "content": system_prompt},
+ {"role": "user", "content": user_content}
+ ]
+
+ # 1. Groq
+ groq_keys = self.asr_engine.groq_keys
+ for model in ["qwen/qwen3.6-27b", "openai/gpt-oss-120b", "openai/gpt-oss-20b", "groq/compound"]:
+ for key in groq_keys:
+ try:
+ res = requests.post(
+ "https://api.groq.com/openai/v1/chat/completions",
+ headers={"Authorization": f"Bearer {key}", "Content-Type": "application/json"},
+ json={"model": model, "messages": messages, "temperature": 0.3, "max_tokens": 4096},
+ timeout=30
+ )
+ if res.status_code == 200:
+ content = res.json()["choices"][0]["message"]["content"].strip()
+ content = re.sub(r".*?", "", content, flags=re.DOTALL).strip()
+ if len(content) > 15:
+ return content
+ except Exception:
+ pass
+
+ # 2. Google Gemini
+ gemini_keys = self.ocr_engine.gemini_keys
+ for model in ["gemini-3.6-flash", "gemini-flash-latest", "gemini-3.5-flash", "gemini-pro-latest"]:
+ for key in gemini_keys:
+ try:
+ url = f"https://generativelanguage.googleapis.com/v1beta/models/{model}:generateContent?key={key}"
+ payload = {
+ "contents": [{"parts": [{"text": f"{system_prompt}\n\n{user_content}"}]}],
+ "generationConfig": {"temperature": 0.3, "maxOutputTokens": 4096}
+ }
+ res = requests.post(url, json=payload, timeout=30)
+ if res.status_code == 200:
+ content = res.json()["candidates"][0]["content"]["parts"][0]["text"].strip()
+ if len(content) > 15:
+ return content
+ except Exception:
+ pass
+
+ # 3. OpenRouter
+ or_keys = self.ocr_engine.openrouter_keys
+ for model in ["nvidia/nemotron-3-super-120b-a12b:free", "nvidia/nemotron-3.5-lightning:free", "openai/gpt-oss-20b:free"]:
+ for key in or_keys:
+ try:
+ res = requests.post(
+ "https://openrouter.ai/api/v1/chat/completions",
+ headers={"Authorization": f"Bearer {key}", "Content-Type": "application/json", "HTTP-Referer": "https://trungsangviet.local", "X-Title": "TrungSangViet"},
+ json={"model": model, "messages": messages, "temperature": 0.3, "max_tokens": 4096},
+ timeout=35
+ )
+ if res.status_code == 200:
+ content = res.json()["choices"][0]["message"]["content"].strip()
+ if len(content) > 15:
+ return content
+ except Exception:
+ pass
+
+ return user_content