Spaces:
Running on Zero
Running on Zero
Fix 2p->20s + 20 bugs: pad audio, OCR/ASR coverage gates, TTS cache/placeholder, omni cloud, ASS escape, timeout, preflight, pool locks, SSRF guards
df03342 verified Download app/core/cloud_pipeline.py from hoangtaiii/DRIPPY4: direct link, hf CLI and curl.
- Browser
- Download file 66.7 kB
-
https://huggingface.co/spaces/hoangtaiii/DRIPPY4/resolve/main/app/core/cloud_pipeline.py
- Command line
-
hf download hf://spaces/hoangtaiii/DRIPPY4/app/core/cloud_pipeline.py
-
curl -L -o cloud_pipeline.py https://huggingface.co/spaces/hoangtaiii/DRIPPY4/resolve/main/app/core/cloud_pipeline.py
66.7 kB
| """ | |
| app/core/cloud_pipeline.py | |
| ────────────────────────── | |
| Pure Cloud-Native Headless Pipeline Coordinator. | |
| Features: | |
| 1. High-Definition Video Preview Frame Generator using FFmpeg (100% Reliable). | |
| 2. Multi-Model ASR Matrix (Groq Whisper Large V3 / OpenRouter / Gemini). | |
| 3. AI Vision Models OCR with automatic ASR fallback. | |
| 4. Active 2026 AI Translation Models (Groq Qwen 3.6 / GPT-OSS 120B / Gemini Flash / Nemotron 3). | |
| 5. Glossary & Slang Mapping (Fashion, Sneaker, Streetwear, Chinese Slang). | |
| 6. 7-Step Quality Guard & Semantic Validation & Timing Compaction. | |
| 7. Subtitle Editing Hooks (extract_subtitles_only & render_from_subtitles). | |
| 8. ASS Subtitle Engine with TikTok / Shorts Typography & Word Jump Styles. | |
| 9. 100% Solid/Delogo Masking Box to eliminate old Chinese subtitles. | |
| 10. Studio SFX & Audio Ducking Mixer. | |
| 11. SQLite JobManager Integration for state persistence. | |
| """ | |
| import os | |
| import re | |
| import sys | |
| import json | |
| import time | |
| import base64 | |
| import shutil | |
| import cv2 | |
| import requests | |
| import subprocess | |
| from pathlib import Path | |
| from typing import Optional, Callable, Dict, Any, Tuple, List | |
| from app.core.cloud_asr import CloudASREngine | |
| from app.core.cloud_ocr import CloudOCREngine | |
| from app.core.cloud_tts import CloudTTSEngine | |
| from app.core.vietnamese_text_normalizer import VietnameseTextNormalizer | |
| from app.core.translation_validator import TranslationValidator | |
| from app.core.translation_quality_guard import TranslationQualityGuard | |
| from app.core.translation_post_editor import post_edit_translation | |
| from app.core.google_translate_fallback import GoogleTranslateFallback | |
| from app.core.subtitle_compactor import compact_blocks_for_tts | |
| from app.core.job_manager import JobManager | |
| from app.core.studio_sfx import generate_sfx_clip | |
| def has_chinese(text: str) -> bool: | |
| """Returns True if string contains CJK Chinese characters.""" | |
| return bool(re.search(r"[\u4e00-\u9fff]", text)) | |
| class CloudPipeline: | |
| def __init__( | |
| self, | |
| base_dir: Optional[Path] = None, | |
| log_callback: Optional[Callable[[str], None]] = None, | |
| progress_callback: Optional[Callable[[int, str], None]] = None, | |
| ffmpeg_path: str = "ffmpeg" | |
| ): | |
| self.base_dir = Path(base_dir or Path(__file__).resolve().parents[2]) | |
| self.output_dir = self.base_dir / "output" | |
| self.temp_dir = self.base_dir / "temp" | |
| self.output_dir.mkdir(parents=True, exist_ok=True) | |
| self.temp_dir.mkdir(parents=True, exist_ok=True) | |
| self.log_fn = log_callback or print | |
| self.progress_fn = progress_callback or (lambda pct, stage: None) | |
| self.ffmpeg_path = ffmpeg_path | |
| self.asr_engine = CloudASREngine(log_fn=self._log) | |
| self.ocr_engine = CloudOCREngine(log_fn=self._log) | |
| self.tts_engine = CloudTTSEngine(log_fn=self._log, ffmpeg_path=self.ffmpeg_path) | |
| self.normalizer = VietnameseTextNormalizer() | |
| self.job_manager = JobManager.instance() | |
| self.glossary = self._load_glossary() | |
| def _log(self, msg: str): | |
| self.log_fn(f"[Cloud Pipeline] {msg}") | |
| def _get_video_duration_sec(self, v_path: Path) -> float: | |
| """Probe video duration (sec). ffprobe -> cv2 fallback. 0.0 if unknown.""" | |
| try: | |
| import shutil as _sh | |
| _ffprobe = _sh.which("ffprobe") | |
| if not _ffprobe: | |
| cand = Path(self.ffmpeg_path).parent / "ffprobe.exe" | |
| _ffprobe = str(cand) if cand.exists() else "ffprobe" | |
| _res = subprocess.run( | |
| [_ffprobe, "-v", "error", "-show_entries", "format=duration", | |
| "-of", "default=noprint_wrappers=1:nokey=1", str(v_path)], | |
| capture_output=True, text=True, timeout=15) | |
| _dur = float((_res.stdout or "").strip()) | |
| if _dur > 0: | |
| return _dur | |
| except Exception: | |
| pass | |
| try: | |
| cap = cv2.VideoCapture(str(v_path)) | |
| fps = cap.get(cv2.CAP_PROP_FPS) or 0 | |
| frames = cap.get(cv2.CAP_PROP_FRAME_COUNT) or 0 | |
| cap.release() | |
| if fps > 0 and frames > 0: | |
| return float(frames / fps) | |
| except Exception: | |
| pass | |
| return 0.0 | |
| def _pad_audio_to_video_duration(self, audio_path: Path, video_dur: float) -> bool: | |
| """Pad (never cut) audio to exactly video duration. Guards against | |
| `-shortest` slicing the video when dubbing/mix is shorter than source | |
| (e.g. mute_original + truncated subtitles: 2min video -> 20s output).""" | |
| if video_dur <= 0 or not audio_path.exists(): | |
| return False | |
| try: | |
| tmp = audio_path.parent / f"{audio_path.stem}_padded.wav" | |
| cmd = [str(self.ffmpeg_path), "-y", "-i", str(audio_path), | |
| "-filter:a", f"apad=whole_dur={video_dur:.3f}", | |
| "-ac", "2", "-ar", "48000", "-t", f"{video_dur:.3f}", str(tmp)] | |
| res = subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.PIPE, timeout=120) | |
| if res.returncode == 0 and tmp.exists() and tmp.stat().st_size > 1000: | |
| tmp.replace(audio_path) | |
| return True | |
| self._log(f"⚠️ Pad audio to {video_dur:.1f}s failed, keeping original mix.") | |
| try: | |
| if tmp.exists(): | |
| tmp.unlink() | |
| except Exception: | |
| pass | |
| except Exception as e: | |
| self._log(f"⚠️ Pad audio error: {e}") | |
| return False | |
| def _load_glossary(self) -> Dict[str, str]: | |
| glossary_path = self.base_dir / "glossary.json" | |
| if glossary_path.exists(): | |
| try: | |
| data = json.loads(glossary_path.read_text(encoding="utf-8")) | |
| return data.get("glossary", {}) | |
| except Exception as e: | |
| self._log(f"⚠️ Lỗi đọc glossary.json: {e}") | |
| return {} | |
| def calculate_region( | |
| self, | |
| video_path: Path, | |
| region_preset: str = "custom", | |
| custom_y_pct: float = 75.0, | |
| custom_h_pct: float = 20.0, | |
| custom_x_pct: float = 0.0, | |
| custom_w_pct: float = 100.0 | |
| ) -> Tuple[int, int, int, int]: | |
| vw, vh = 1920, 1080 | |
| try: | |
| cap = cv2.VideoCapture(str(video_path)) | |
| w = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH)) | |
| h = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT)) | |
| cap.release() | |
| if w > 0 and h > 0: | |
| vw, vh = w, h | |
| except Exception: | |
| pass | |
| if region_preset == "bottom_25": | |
| x = 0 | |
| w = vw | |
| y = int(vh * 0.72) | |
| h = int(vh * 0.24) | |
| elif region_preset == "bottom_15": | |
| x = 0 | |
| w = vw | |
| y = int(vh * 0.82) | |
| h = int(vh * 0.15) | |
| elif region_preset == "middle_25": | |
| x = 0 | |
| w = vw | |
| y = int(vh * 0.38) | |
| h = int(vh * 0.24) | |
| elif region_preset == "none": | |
| return (0, 0, 0, 0) | |
| elif region_preset == "custom": | |
| x = int(vw * (custom_x_pct / 100.0)) | |
| w = int(vw * (custom_w_pct / 100.0)) | |
| y = int(vh * (custom_y_pct / 100.0)) | |
| h = int(vh * (custom_h_pct / 100.0)) | |
| else: | |
| return (0, 0, 0, 0) | |
| x = max(0, min(vw - 2, x)) | |
| y = max(0, min(vh - 2, y)) | |
| w = max(2, min(vw - x, w)) | |
| h = max(2, min(vh - y, h)) | |
| return (x, y, w, h) | |
| def generate_preview_frame( | |
| self, | |
| video_path: str, | |
| region_preset: str = "custom", | |
| custom_y_pct: float = 75.0, | |
| custom_h_pct: float = 20.0, | |
| custom_x_pct: float = 0.0, | |
| custom_w_pct: float = 100.0, | |
| sec: float = 2.0, | |
| return_type: str = "rgb", | |
| # ── Logo overlay params ── | |
| logo_enabled: bool = False, | |
| logo_path: str = "", | |
| logo_preset: str = "bottom_right", | |
| logo_scale: float = 15.0, | |
| logo_opacity: float = 1.0, | |
| logo_x_pct: float = 80.0, | |
| logo_y_pct: float = 80.0, | |
| logo_chromakey: bool = True, | |
| # ── CTA video overlay params (appears every 5s) ── | |
| cta_enabled: bool = False, | |
| cta_path: str = "", | |
| cta_preset: str = "bottom_center", | |
| cta_scale: float = 35.0, | |
| cta_opacity: float = 1.0, | |
| cta_x_pct: float = 50.0, | |
| cta_y_pct: float = 85.0, | |
| cta_interval: float = 5.0, | |
| cta_duration: float = 2.0, | |
| cta_chromakey: bool = True | |
| ): | |
| """Extracts a frame at timestamp sec and overlays bounding box + logo + CTA preview.""" | |
| v_file = Path(video_path) | |
| if not v_file.exists(): | |
| return None | |
| raw_frame_path = self.temp_dir / f"raw_frame_{int(time.time()*1000)}.jpg" | |
| cmd = [ | |
| str(self.ffmpeg_path), "-ss", str(sec), "-y", | |
| "-i", str(v_file), | |
| "-frames:v", "1", "-q:v", "2", | |
| str(raw_frame_path) | |
| ] | |
| try: | |
| subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True) | |
| frame = cv2.imread(str(raw_frame_path)) | |
| if raw_frame_path.exists(): | |
| raw_frame_path.unlink() | |
| except Exception: | |
| frame = None | |
| if frame is None: | |
| cap = cv2.VideoCapture(str(v_file)) | |
| fps = cap.get(cv2.CAP_PROP_FPS) or 25.0 | |
| target_frame = max(0, int(sec * fps)) | |
| cap.set(cv2.CAP_PROP_POS_FRAMES, target_frame) | |
| ret, frame = cap.read() | |
| cap.release() | |
| if frame is None: | |
| return None | |
| x, y, w, h = self.calculate_region( | |
| v_file, region_preset, custom_y_pct, custom_h_pct, custom_x_pct, custom_w_pct | |
| ) | |
| if w > 0 and h > 0: | |
| overlay = frame.copy() | |
| cv2.rectangle(overlay, (x, y), (x + w, y + h), (0, 0, 255), -1) | |
| cv2.addWeighted(overlay, 0.35, frame, 0.65, 0, frame) | |
| cv2.rectangle(frame, (x, y), (x + w, y + h), (0, 255, 255), 3) | |
| label = "VUNG CHE SUB CU & QUET OCR" | |
| font = cv2.FONT_HERSHEY_SIMPLEX | |
| font_scale = max(0.5, frame.shape[1] / 1200.0) | |
| thickness = 2 | |
| (tw, th), _ = cv2.getTextSize(label, font, font_scale, thickness) | |
| label_y = max(th + 10, y - 8) | |
| cv2.rectangle(frame, (x, label_y - th - 6), (x + tw + 10, label_y + 4), (0, 0, 0), -1) | |
| cv2.putText(frame, label, (x + 5, label_y), font, font_scale, (0, 255, 255), thickness, cv2.LINE_AA) | |
| # ── Logo overlay preview ── | |
| if logo_enabled and logo_path and Path(logo_path).exists(): | |
| try: | |
| from app.core.logo_overlay import load_logo_rgba, overlay_logo_on_frame | |
| _logo_rgba = load_logo_rgba(logo_path, chromakey=logo_chromakey) | |
| if _logo_rgba is not None: | |
| frame = overlay_logo_on_frame( | |
| frame_bgr=frame, | |
| logo_rgba=_logo_rgba, | |
| preset=logo_preset, | |
| scale_pct=float(logo_scale), | |
| opacity=float(logo_opacity), | |
| custom_x_pct=float(logo_x_pct), | |
| custom_y_pct=float(logo_y_pct), | |
| margin_pct=2.0 | |
| ) | |
| cv2.putText(frame, f"LOGO:{logo_preset} {logo_scale:.0f}%", (10, 30), cv2.FONT_HERSHEY_SIMPLEX, 0.7, (0, 255, 255), 2, cv2.LINE_AA) | |
| except Exception as _e: | |
| self._log(f"⚠️ Logo preview error: {_e}") | |
| # ── CTA video overlay preview (every 5s) ── | |
| if cta_enabled and cta_path and Path(cta_path).exists(): | |
| try: | |
| from app.core.logo_overlay import load_cta_frame_rgba, overlay_cta_on_frame, should_show_cta_at_time | |
| if should_show_cta_at_time(float(sec), float(cta_interval), float(cta_duration)): | |
| # CTA time loops within CTA video duration | |
| _cta_time = float(sec) % float(cta_interval) | |
| # If CTA video is shorter than interval, loop inside | |
| _cta_rgba = load_cta_frame_rgba(cta_path, cta_time_sec=_cta_time, chromakey=cta_chromakey) | |
| if _cta_rgba is not None: | |
| frame = overlay_cta_on_frame( | |
| frame_bgr=frame, | |
| cta_rgba=_cta_rgba, | |
| preset=cta_preset, | |
| scale_pct=float(cta_scale), | |
| opacity=float(cta_opacity), | |
| custom_x_pct=float(cta_x_pct), | |
| custom_y_pct=float(cta_y_pct), | |
| margin_pct=2.0 | |
| ) | |
| cv2.putText(frame, f"CTA:{cta_preset} {cta_scale:.0f}% every {cta_interval:.0f}s", (10, 60), cv2.FONT_HERSHEY_SIMPLEX, 0.6, (0, 255, 0), 2, cv2.LINE_AA) | |
| else: | |
| cv2.putText(frame, f"CTA: hidden at {sec:.1f}s (interval {cta_interval:.0f}s)", (10, 60), cv2.FONT_HERSHEY_SIMPLEX, 0.5, (0, 165, 255), 1, cv2.LINE_AA) | |
| except Exception as _e: | |
| self._log(f"⚠️ CTA preview error: {_e}") | |
| if return_type == "base64": | |
| _, buffer = cv2.imencode('.jpg', frame, [cv2.IMWRITE_JPEG_QUALITY, 85]) | |
| return base64.b64encode(buffer).decode('utf-8') | |
| return cv2.cvtColor(frame, cv2.COLOR_BGR2RGB) | |
| def extract_subtitles_only( | |
| self, | |
| video_path: str, | |
| mode: str = "asr", | |
| source_lang: str = "zh", | |
| region_preset: str = "custom", | |
| custom_y_pct: float = 75.0, | |
| custom_h_pct: float = 20.0, | |
| custom_x_pct: float = 0.0, | |
| custom_w_pct: float = 100.0 | |
| ) -> Optional[Dict[str, Any]]: | |
| """ | |
| Runs Stage 0 -> Stage A -> Stage C. | |
| Returns parsed subtitle blocks (original & translated) for Web Subtitle Editor. | |
| """ | |
| v_path = Path(video_path) | |
| if not v_path.exists(): | |
| self._log(f"❌ Video not found: {video_path}") | |
| return None | |
| video_stem = v_path.stem | |
| vtd = self.temp_dir / video_stem | |
| vtd.mkdir(parents=True, exist_ok=True) | |
| self.progress_fn(5, "STAGE_0_PREPARE") | |
| calc_region = self.calculate_region( | |
| v_path, region_preset, custom_y_pct, custom_h_pct, custom_x_pct, custom_w_pct | |
| ) | |
| x, y, w, h = calc_region | |
| # 1. Extract audio | |
| extracted_audio = vtd / "extracted_audio.wav" | |
| self._log("⚡ Trích xuất âm thanh từ video gốc...") | |
| cmd = [ | |
| str(self.ffmpeg_path), "-y", "-i", str(v_path), | |
| "-vn", "-acodec", "pcm_s16le", "-ar", "16000", "-ac", "1", | |
| str(extracted_audio) | |
| ] | |
| subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True) | |
| # 2. OCR / ASR | |
| self.progress_fn(25, "STAGE_A_OCR_ASR") | |
| original_srt = vtd / "original.srt" | |
| if mode == "ocr": | |
| lang_label = "Tiếng Trung 🇨🇳" if str(source_lang).lower().startswith("zh") else "Tiếng Anh 🇺🇸" | |
| self._log(f"👁️ Quét chữ phụ đề bằng AI Vision Models (ưu tiên {lang_label} - source_lang={source_lang})...") | |
| ocr_box = (x, y, w, h) if (w > 0 and h > 0) else None | |
| ok = self.ocr_engine.scan_video_subtitles_to_srt(str(v_path), str(original_srt), blur_region=ocr_box, source_lang=source_lang) | |
| if not ok or not original_srt.exists() or original_srt.stat().st_size < 10: | |
| # Report detailed reason before fallback | |
| reason = [] | |
| if not ok: | |
| reason.append("Vision API trả về rỗng / lỗi key") | |
| if not original_srt.exists(): | |
| reason.append("SRT file không tồn tại") | |
| elif original_srt.stat().st_size < 10: | |
| reason.append(f"SRT quá nhỏ ({original_srt.stat().st_size} bytes)") | |
| self._log(f"⚠️ OCR không phát hiện chữ ({', '.join(reason)}) -> Tự động chuyển sang Cloud ASR (fallback)...") | |
| self._log(f" 💡 Kiểm tra: 1) API Key Gemini/OpenRouter còn hạn? 2) Vùng cắt [{x},{y},{w},{h}] có đúng chứa sub? 3) Thử tăng độ nhạy hoặc chọn preset Đáy Màn Hình") | |
| ok = self.asr_engine.transcribe_audio_to_srt(str(extracted_audio), str(original_srt), source_lang=source_lang) | |
| else: | |
| self._log("🎙️ Nhận diện giọng nói bằng Cloud ASR (Groq Whisper Large-v3)...") | |
| ok = self.asr_engine.transcribe_audio_to_srt(str(extracted_audio), str(original_srt), source_lang=source_lang) | |
| if not ok or not original_srt.exists() or original_srt.stat().st_size < 10: | |
| raise RuntimeError("Không thể trích xuất phụ đề từ video.") | |
| # 3. Translation | |
| self.progress_fn(50, "STAGE_C_TRANSLATION") | |
| self._log("🌐 Dịch thuật bằng AI Translation Matrix...") | |
| translated_srt = vtd / "translated.srt" | |
| self._translate_srt_cloud(original_srt, translated_srt, source_lang=source_lang) | |
| orig_blocks = self._parse_srt_blocks(original_srt.read_text(encoding="utf-8", errors="ignore")) | |
| trans_blocks = self._parse_srt_blocks(translated_srt.read_text(encoding="utf-8", errors="ignore")) | |
| combined = [] | |
| for ob in orig_blocks: | |
| tb = next((t for t in trans_blocks if t["id"] == ob["id"]), None) | |
| combined.append({ | |
| "id": ob["id"], | |
| "timing": ob["timing"], | |
| "start_ms": ob["start_ms"], | |
| "end_ms": ob["end_ms"], | |
| "original_text": ob["text"], | |
| "vietnamese_text": tb["text"] if tb else ob["text"], | |
| "sfx": "" | |
| }) | |
| return { | |
| "video_stem": video_stem, | |
| "blocks": combined, | |
| "original_srt": str(original_srt), | |
| "translated_srt": str(translated_srt) | |
| } | |
| def render_from_subtitles( | |
| self, | |
| video_path: str, | |
| subtitles_data: List[Dict[str, Any]], | |
| voice: str = "vi-VN-NamMinhNeural", | |
| speed: float = 1.0, | |
| pitch: int = 0, | |
| volume: int = 100, | |
| region_preset: str = "custom", | |
| sub_mask_mode: str = "box", | |
| sub_style: str = "motion_drip", | |
| custom_y_pct: float = 75.0, | |
| custom_h_pct: float = 20.0, | |
| custom_x_pct: float = 0.0, | |
| custom_w_pct: float = 100.0, | |
| ducking_ratio: float = 0.18, | |
| enable_sfx: bool = True, | |
| mute_original_audio: bool = False, | |
| # ── Logo overlay ── | |
| logo_enabled: bool = False, | |
| logo_path: str = "", | |
| logo_preset: str = "bottom_right", | |
| logo_scale: float = 15.0, | |
| logo_opacity: float = 1.0, | |
| logo_x_pct: float = 80.0, | |
| logo_y_pct: float = 80.0, | |
| logo_chromakey: bool = True, | |
| # ── CTA video overlay (every 5s) ── | |
| cta_enabled: bool = False, | |
| cta_path: str = "", | |
| cta_preset: str = "bottom_center", | |
| cta_scale: float = 35.0, | |
| cta_opacity: float = 1.0, | |
| cta_x_pct: float = 50.0, | |
| cta_y_pct: float = 85.0, | |
| cta_interval: float = 5.0, | |
| cta_duration: float = 2.0, | |
| cta_chromakey: bool = True | |
| ) -> Optional[str]: | |
| """Completes TTS synthesis, Ducking audio mix, and video rendering from user-reviewed subtitles.""" | |
| v_path = Path(video_path) | |
| if not v_path.exists(): | |
| return None | |
| video_stem = v_path.stem | |
| vtd = self.temp_dir / video_stem | |
| vtd.mkdir(parents=True, exist_ok=True) | |
| start_time = time.time() | |
| calc_region = self.calculate_region( | |
| v_path, region_preset, custom_y_pct, custom_h_pct, custom_x_pct, custom_w_pct | |
| ) | |
| x, y, w, h = calc_region | |
| # Write edited SRT | |
| translated_srt = vtd / "translated.srt" | |
| srt_lines = [] | |
| for item in subtitles_data: | |
| srt_lines.append(str(item["id"])) | |
| srt_lines.append(item["timing"]) | |
| vi_text = self.normalizer.normalize(item["vietnamese_text"]) if hasattr(self.normalizer, "normalize") else item["vietnamese_text"] | |
| srt_lines.append(vi_text) | |
| srt_lines.append("") | |
| translated_srt.write_text("\n".join(srt_lines), encoding="utf-8") | |
| # 4. Cloud TTS | |
| self.progress_fn(65, "STAGE_TTS_DUBBING") | |
| self._log(f"🎙️ Tạo giọng đọc lồng tiếng ({voice}, speed={speed}x, volume={volume}%)...") | |
| dubbing_wav = vtd / "dubbing.wav" | |
| ok_tts = self.tts_engine.synthesize_srt_to_audio( | |
| str(translated_srt), | |
| str(dubbing_wav), | |
| voice=voice, | |
| speed=speed, | |
| pitch=pitch, | |
| volume=volume, | |
| temp_dir=str(vtd / "tts_segments") | |
| ) | |
| if not ok_tts or not dubbing_wav.exists(): | |
| raise RuntimeError("Lỗi tạo giọng đọc TTS.") | |
| # 5. SFX & Audio Ducking | |
| self.progress_fn(80, "STAGE_D_AUDIO_MIX") | |
| extracted_audio = vtd / "extracted_audio.wav" | |
| if not extracted_audio.exists(): | |
| cmd = [ | |
| str(self.ffmpeg_path), "-y", "-i", str(v_path), | |
| "-vn", "-acodec", "pcm_s16le", "-ar", "16000", "-ac", "1", | |
| str(extracted_audio) | |
| ] | |
| subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True) | |
| mixed_audio = vtd / "mixed_final.wav" | |
| video_dur_sec = self._get_video_duration_sec(v_path) | |
| if video_dur_sec > 0: | |
| self._log(f"⏱️ Video gốc dài {video_dur_sec:.1f}s — audio mix sẽ được pad đúng bằng, chống cắt ngọn bởi -shortest.") | |
| if mute_original_audio: | |
| self._log("🔇 Tắt hoàn toàn tiếng gốc — chỉ giữ giọng Việt (mute_original_audio=ON)") | |
| # Dubbing wav đã được time-stretch canvas đúng duration, chỉ cần chuẩn hóa sample-rate | |
| cmd_mix = [ | |
| str(self.ffmpeg_path), "-y", | |
| "-i", str(dubbing_wav), | |
| "-ac", "2", "-ar", "48000", | |
| str(mixed_audio) | |
| ] | |
| subprocess.run(cmd_mix, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True) | |
| # FIX 2p->20s: pad im lặng tới đúng video duration để -shortest không cắt video | |
| if video_dur_sec > 0: | |
| self._pad_audio_to_video_duration(mixed_audio, video_dur_sec) | |
| else: | |
| self._log("🎚️ Trộn nhạc nền (Auto Ducking) + Giọng đọc thuyết minh...") | |
| filter_complex = ( | |
| f"[0:a]volume={ducking_ratio}[bg];" | |
| f"[1:a]volume=1.0[dub];" | |
| f"[bg][dub]amix=inputs=2:duration=longest:dropout_transition=2:normalize=0[aout]" | |
| ) | |
| cmd_mix = [ | |
| str(self.ffmpeg_path), "-y", | |
| "-i", str(extracted_audio), | |
| "-i", str(dubbing_wav), | |
| "-filter_complex", filter_complex, | |
| "-map", "[aout]", | |
| "-ac", "2", "-ar", "48000", | |
| str(mixed_audio) | |
| ] | |
| subprocess.run(cmd_mix, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True) | |
| # Safety: amix duration=longest lẽ ra đã full, nhưng pad lại cho chắc | |
| if video_dur_sec > 0: | |
| self._pad_audio_to_video_duration(mixed_audio, video_dur_sec) | |
| # 6. Render Video | |
| self.progress_fn(90, "STAGE_D_RENDER") | |
| self._log("🎬 Render video hoàn thiện: Xóa sub cũ & Đè sub tiếng Việt...") | |
| final_output = self.output_dir / f"studio_final_{v_path.name}" | |
| # Probe dimensions with cv2 then ffprobe fallback (handles vertical & cv2-missing envs) | |
| vw, vh = 1920, 1080 | |
| try: | |
| cap = cv2.VideoCapture(str(v_path)) | |
| w_tmp = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH)) or 0 | |
| h_tmp = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT)) or 0 | |
| cap.release() | |
| if w_tmp > 0 and h_tmp > 0: | |
| vw, vh = w_tmp, h_tmp | |
| else: | |
| raise ValueError("cv2 returned 0") | |
| except Exception: | |
| try: | |
| import shutil as _sh | |
| _ffprobe = _sh.which("ffprobe") or str(Path(self.ffmpeg_path).parent / "ffprobe.exe") if Path(self.ffmpeg_path).exists() else "ffprobe" | |
| if not Path(_ffprobe).exists(): | |
| _ffprobe = "ffprobe" | |
| _res = subprocess.run([_ffprobe, "-v", "error", "-select_streams", "v:0", "-show_entries", "stream=width,height", "-of", "json", str(v_path)], capture_output=True, text=True, timeout=5) | |
| _j = json.loads(_res.stdout or "{}") | |
| _ws = _j.get("streams", [{}])[0] | |
| if _ws.get("width") and _ws.get("height"): | |
| vw, vh = int(_ws["width"]), int(_ws["height"]) | |
| except Exception: | |
| pass | |
| translated_ass = vtd / "translated.ass" | |
| self._convert_srt_to_ass(translated_srt, translated_ass, vw, vh, x, y, w, h, sub_style) | |
| ass_escaped = str(translated_ass).replace("\\", "/").replace(":", "\\:") | |
| sub_filter = f"subtitles='{ass_escaped}'" | |
| vf_filters = [] | |
| if w > 0 and h > 0: | |
| if sub_mask_mode == "delogo": | |
| safe_x = max(2, x) | |
| safe_y = max(2, y) | |
| safe_w = max(4, min(w, vw - safe_x - 2)) | |
| safe_h = max(4, min(h, vh - safe_y - 2)) | |
| vf_filters.append(f"delogo=x={safe_x}:y={safe_y}:w={safe_w}:h={safe_h}") | |
| elif sub_mask_mode == "box": | |
| vf_filters.append(f"drawbox=x={x}:y={y}:w={w}:h={h}:color=black@0.92:t=fill") | |
| # Determine logo & CTA overlay enabled (with fallback to bundled assets) | |
| _logo_enabled = bool(logo_enabled and logo_path and Path(logo_path).exists()) | |
| if logo_enabled and not _logo_enabled: | |
| _fb = self.base_dir / "assets" / "logo_ins_drippy4.png" | |
| if _fb.exists(): | |
| _logo_enabled = True | |
| logo_path = str(_fb) | |
| _logo_path = Path(logo_path) if _logo_enabled else None | |
| _cta_enabled = bool(cta_enabled and cta_path and Path(cta_path).exists()) | |
| if cta_enabled and not _cta_enabled: | |
| _fb2 = self.base_dir / "assets" / "cta_ins_drippy4.mp4" | |
| if _fb2.exists(): | |
| _cta_enabled = True | |
| cta_path = str(_fb2) | |
| _cta_path = Path(cta_path) if _cta_enabled else None | |
| # If any overlay enabled, we need filter_complex | |
| if _logo_enabled or _cta_enabled: | |
| if _logo_enabled: | |
| self._log(f"🖼️ Overlay logo: {Path(logo_path).name} preset={logo_preset} scale={logo_scale}% opacity={logo_opacity} chromakey={logo_chromakey}") | |
| if _cta_enabled: | |
| self._log(f"🎬 Overlay CTA: {Path(cta_path).name} preset={cta_preset} scale={cta_scale}% every {cta_interval}s for {cta_duration}s chromakey={cta_chromakey}") | |
| # Build filter parts step-by-step. Inputs: 0:video, 1:audio, 2:logo(if), 3:cta(if) | |
| # Determine indices | |
| _logo_idx = 2 if _logo_enabled else None | |
| _cta_idx = None | |
| if _cta_enabled: | |
| _cta_idx = 3 if _logo_enabled else 2 | |
| # Video preprocessing (delogo/box) -> [base] | |
| _filter_parts = [] | |
| if vf_filters and len([f for f in vf_filters if f != sub_filter]) > 0: | |
| _pre_vf = ",".join([f for f in vf_filters if f != sub_filter]) | |
| _filter_parts.append(f"[0:v]{_pre_vf}[base]") | |
| _cur = "base" | |
| else: | |
| _filter_parts.append("[0:v]null[base]") | |
| _cur = "base" | |
| # Logo overlay | |
| if _logo_enabled: | |
| _logo_w_orig, _logo_h_orig = 2400, 1792 | |
| try: | |
| _t = cv2.imread(str(_logo_path), cv2.IMREAD_UNCHANGED) | |
| if _t is not None: | |
| _logo_h_orig, _logo_w_orig = _t.shape[:2] | |
| except Exception: | |
| pass | |
| _target_w = vw * float(logo_scale) / 100.0 | |
| _sf = max(0.02, min(0.5, _target_w / float(max(1, _logo_w_orig)))) | |
| _logo_vf_parts = [] | |
| if logo_chromakey: | |
| _logo_vf_parts.append("colorkey=0x00FF00:0.3:0.1") | |
| _logo_vf_parts.append("format=rgba") | |
| _logo_vf_parts.append(f"scale=iw*{_sf:.4f}:ih*{_sf:.4f}:flags=lanczos") | |
| if float(logo_opacity) < 0.99: | |
| _logo_vf_parts.append(f"colorchannelmixer=aa={float(logo_opacity):.2f}") | |
| _logo_vf = ",".join(_logo_vf_parts) | |
| _m = 2.0 | |
| if logo_preset == "top_left": | |
| _lx, _ly = f"W*{_m/100:.3f}", f"H*{_m/100:.3f}" | |
| elif logo_preset == "top_right": | |
| _lx, _ly = f"W-w-W*{_m/100:.3f}", f"H*{_m/100:.3f}" | |
| elif logo_preset == "bottom_left": | |
| _lx, _ly = f"W*{_m/100:.3f}", f"H-h-H*{_m/100:.3f}" | |
| elif logo_preset == "bottom_right": | |
| _lx, _ly = f"W-w-W*{_m/100:.3f}", f"H-h-H*{_m/100:.3f}" | |
| elif logo_preset == "center": | |
| _lx, _ly = "(W-w)/2", "(H-h)/2" | |
| elif logo_preset == "top_center": | |
| _lx, _ly = "(W-w)/2", f"H*{_m/100:.3f}" | |
| elif logo_preset == "bottom_center": | |
| _lx, _ly = "(W-w)/2", f"H-h-H*{_m/100:.3f}" | |
| elif logo_preset == "custom": | |
| _lx, _ly = f"W*{float(logo_x_pct)/100:.4f}", f"H*{float(logo_y_pct)/100:.4f}" | |
| else: | |
| _lx, _ly = f"W-w-W*{_m/100:.3f}", f"H-h-H*{_m/100:.3f}" | |
| _filter_parts.append(f"[{_logo_idx}:v]{_logo_vf}[logo]") | |
| _next = "with_logo" if _cta_enabled or sub_filter else "v" | |
| _filter_parts.append(f"[{_cur}][logo]overlay={_lx}:{_ly}:format=rgb[{_next}]") | |
| _cur = _next | |
| # CTA overlay (periodic every interval) | |
| if _cta_enabled: | |
| # Probe CTA size | |
| _cta_w_orig, _cta_h_orig = 1080, 1920 | |
| try: | |
| _cap = cv2.VideoCapture(str(_cta_path)) | |
| _cta_w_orig = int(_cap.get(cv2.CAP_PROP_FRAME_WIDTH)) or _cta_w_orig | |
| _cta_h_orig = int(_cap.get(cv2.CAP_PROP_FRAME_HEIGHT)) or _cta_h_orig | |
| _cap.release() | |
| except Exception: | |
| pass | |
| # TikTok CTA bump: enforce minimum 42% width for mobile visibility | |
| effective_cta_scale = max(float(cta_scale), 42.0) | |
| _cta_target_w = vw * effective_cta_scale / 100.0 | |
| _cta_sf = max(0.05, min(1.0, _cta_target_w / float(max(1, _cta_w_orig)))) | |
| # Clamp height to 80% of video height to avoid overflow (CTA vertical on horizontal video) | |
| _cta_target_h = _cta_h_orig * _cta_sf | |
| _max_h = vh * 0.90 | |
| if _cta_target_h > _max_h: | |
| _cta_sf = _max_h / float(max(1, _cta_h_orig)) | |
| _cta_target_w = _cta_w_orig * _cta_sf | |
| _cta_vf_parts = [] | |
| if cta_chromakey: | |
| _cta_vf_parts.append("colorkey=0x00FF00:0.3:0.1") | |
| _cta_vf_parts.append("format=rgba") | |
| _cta_vf_parts.append(f"scale=iw*{_cta_sf:.4f}:ih*{_cta_sf:.4f}:flags=lanczos") | |
| if float(cta_opacity) < 0.99: | |
| _cta_vf_parts.append(f"colorchannelmixer=aa={float(cta_opacity):.2f}") | |
| _cta_vf = ",".join(_cta_vf_parts) | |
| _m2 = 2.0 | |
| if cta_preset == "top_left": | |
| _cx, _cy = f"W*{_m2/100:.3f}", f"H*{_m2/100:.3f}" | |
| elif cta_preset == "top_right": | |
| _cx, _cy = f"W-w-W*{_m2/100:.3f}", f"H*{_m2/100:.3f}" | |
| elif cta_preset == "bottom_left": | |
| _cx, _cy = f"W*{_m2/100:.3f}", f"H-h-H*{_m2/100:.3f}" | |
| elif cta_preset == "bottom_right": | |
| _cx, _cy = f"W-w-W*{_m2/100:.3f}", f"H-h-H*{_m2/100:.3f}" | |
| elif cta_preset == "center": | |
| _cx, _cy = "(W-w)/2", "(H-h)/2" | |
| elif cta_preset == "top_center": | |
| _cx, _cy = "(W-w)/2", f"H*{_m2/100:.3f}" | |
| elif cta_preset == "bottom_center": | |
| _cx, _cy = "(W-w)/2", f"H-h-H*{_m2/100:.3f}" | |
| elif cta_preset == "custom": | |
| _cx, _cy = f"W*{float(cta_x_pct)/100:.4f}", f"H*{float(cta_y_pct)/100:.4f}" | |
| else: | |
| _cx, _cy = "(W-w)/2", f"H-h-H*{_m2/100:.3f}" | |
| _enable = f"lt(mod(t\\,{float(cta_interval)})\\,{float(cta_duration)})" | |
| _filter_parts.append(f"[{_cta_idx}:v]{_cta_vf}[cta]") | |
| _next2 = "v" if not sub_filter else "with_cta" | |
| # Use escaped comma for FFmpeg enable expression | |
| _filter_parts.append(f"[{_cur}][cta]overlay={_cx}:{_cy}:format=rgb:enable='{_enable}'[{_next2}]") | |
| _cur = _next2 | |
| # TikTok polish: CFR 30 + even dims + light sharpen BEFORE subtitles (keeps text razor sharp) | |
| tiktok_polish = "fps=30:round=near,scale=trunc(iw/2)*2:trunc(ih/2)*2:flags=lanczos+accurate_rnd+full_chroma_int:sws_dither=ed,unsharp=3:3:0.35:3:3:0.0" | |
| if sub_filter: | |
| _filter_parts.append(f"[{_cur}]{tiktok_polish}[polished]") | |
| _cur = "polished" | |
| _filter_parts.append(f"[{_cur}]{sub_filter}[v]") | |
| _cur = "v" | |
| else: | |
| _filter_parts.append(f"[{_cur}]{tiktok_polish}[v]") | |
| _cur = "v" | |
| _filter_complex = ";".join(_filter_parts) | |
| # Build ffmpeg inputs: video, audio, logo(if), cta(if) - add -shortest when looped overlays present | |
| cmd_inputs = [str(self.ffmpeg_path), "-y", "-fflags", "+genpts", "-avoid_negative_ts", "make_zero", "-i", str(v_path), "-i", str(mixed_audio)] | |
| if _logo_enabled: | |
| cmd_inputs += ["-loop", "1", "-i", str(_logo_path)] | |
| if _cta_enabled: | |
| cmd_inputs += ["-stream_loop", "999", "-i", str(_cta_path)] | |
| _extra = ["-shortest"] if (_logo_enabled or _cta_enabled) else [] | |
| # TikTok spec: High Profile yuv420p, 30 CFR, 48k AAC, faststart, accurate sync | |
| tiktok_video_args = ["-r", "30", "-c:v", "libx264", "-preset", "medium", "-crf", "18", "-profile:v", "high", "-level", "4.1", "-pix_fmt", "yuv420p", "-g", "60", "-keyint_min", "30", "-sc_threshold", "0", "-x264-params", "ref=4:bframes=2:me=hex:subme=7:psy=1:psy-rd=0.8:aq-mode=2", "-colorspace", "bt709", "-color_primaries", "bt709", "-color_trc", "bt709", "-color_range", "tv"] | |
| tiktok_audio_args = ["-c:a", "aac", "-profile:a", "aac_low", "-ar", "48000", "-ac", "2", "-b:a", "192k", "-af", "aresample=async=1:min_hard_comp=0.100000:first_pts=0"] | |
| tiktok_mux_args = ["-movflags", "+faststart", "-fflags", "+genpts", "-max_interleave_delta", "100M", "-vsync", "cfr", "-fps_mode", "cfr"] | |
| cmd_render = cmd_inputs + ["-filter_complex", _filter_complex, "-map", "[v]", "-map", "1:a:0"] + _extra + tiktok_video_args + tiktok_audio_args + tiktok_mux_args + ["-shortest", str(final_output)] | |
| res = subprocess.run(cmd_render, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True) | |
| if res.returncode != 0: | |
| self._log(f"⚠️ Overlay render failed ({res.stderr[:300]}), fallback to TikTok-spec normal render...") | |
| # Rebuild VF with TikTok polish inserted before subs (same as non-overlay branch) | |
| tiktok_polish_fb = "fps=30:round=near,scale=trunc(iw/2)*2:trunc(ih/2)*2:flags=lanczos+accurate_rnd+full_chroma_int:sws_dither=ed,unsharp=3:3:0.35:3:3:0.0" | |
| vf_without_sub_fb = [f for f in vf_filters if f != sub_filter] | |
| ordered_fb = [] | |
| if vf_without_sub_fb: | |
| ordered_fb.extend(vf_without_sub_fb) | |
| ordered_fb.append(tiktok_polish_fb) | |
| ordered_fb.append(sub_filter) | |
| final_vf = ",".join(ordered_fb) | |
| cmd_render = [ | |
| str(self.ffmpeg_path), "-y", | |
| "-fflags", "+genpts", "-avoid_negative_ts", "make_zero", | |
| "-i", str(v_path), | |
| "-i", str(mixed_audio), | |
| "-vf", final_vf, | |
| "-map", "0:v:0", "-map", "1:a:0", | |
| "-r", "30", | |
| "-c:v", "libx264", "-preset", "medium", "-crf", "18", | |
| "-profile:v", "high", "-level", "4.1", "-pix_fmt", "yuv420p", | |
| "-g", "60", "-keyint_min", "30", "-sc_threshold", "0", | |
| "-x264-params", "ref=4:bframes=2:me=hex:subme=7:psy=1:psy-rd=0.8:aq-mode=2", | |
| "-colorspace", "bt709", "-color_primaries", "bt709", "-color_trc", "bt709", "-color_range", "tv", | |
| "-c:a", "aac", "-profile:a", "aac_low", "-ar", "48000", "-ac", "2", "-b:a", "192k", | |
| "-af", "aresample=async=1:min_hard_comp=0.100000:first_pts=0", | |
| "-movflags", "+faststart", | |
| "-fflags", "+genpts", "-max_interleave_delta", "100M", | |
| "-vsync", "cfr", "-fps_mode", "cfr", | |
| "-shortest", | |
| str(final_output) | |
| ] | |
| res = subprocess.run(cmd_render, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True) | |
| if res.returncode != 0: | |
| self._log(f"⚠️ Fallback render không mask: {res.stderr[:120]}") | |
| fallback_cmd = [ | |
| str(self.ffmpeg_path), "-y", | |
| "-fflags", "+genpts", | |
| "-i", str(v_path), | |
| "-i", str(mixed_audio), | |
| "-vf", f"{tiktok_polish_fb},{sub_filter}", | |
| "-map", "0:v:0", "-map", "1:a:0", | |
| "-r", "30", | |
| "-c:v", "libx264", "-preset", "medium", "-crf", "18", | |
| "-profile:v", "high", "-pix_fmt", "yuv420p", | |
| "-c:a", "aac", "-profile:a", "aac_low", "-ar", "48000", "-ac", "2", "-b:a", "192k", | |
| "-movflags", "+faststart", | |
| "-shortest", | |
| str(final_output) | |
| ] | |
| subprocess.run(fallback_cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True) | |
| else: | |
| # TikTok polish inserted before subtitles for sharpness + CFR + even dims | |
| tiktok_polish = "fps=30:round=near,scale=trunc(iw/2)*2:trunc(ih/2)*2:flags=lanczos+accurate_rnd+full_chroma_int:sws_dither=ed,unsharp=3:3:0.35:3:3:0.0" | |
| # Order: mask (delogo/box) -> polish -> subtitles | |
| vf_without_sub = [f for f in vf_filters if f != sub_filter] | |
| # Rebuild vf in TikTok-optimal order | |
| ordered_vf = [] | |
| if vf_without_sub: | |
| ordered_vf.extend(vf_without_sub) | |
| ordered_vf.append(tiktok_polish) | |
| ordered_vf.append(sub_filter) | |
| final_vf = ",".join(ordered_vf) | |
| cmd_render = [ | |
| str(self.ffmpeg_path), "-y", | |
| "-fflags", "+genpts", "-avoid_negative_ts", "make_zero", | |
| "-i", str(v_path), | |
| "-i", str(mixed_audio), | |
| "-vf", final_vf, | |
| "-map", "0:v:0", "-map", "1:a:0", | |
| "-r", "30", | |
| "-c:v", "libx264", "-preset", "medium", "-crf", "18", | |
| "-profile:v", "high", "-level", "4.1", "-pix_fmt", "yuv420p", | |
| "-g", "60", "-keyint_min", "30", "-sc_threshold", "0", | |
| "-x264-params", "ref=4:bframes=2:me=hex:subme=7:psy=1:psy-rd=0.8:aq-mode=2", | |
| "-colorspace", "bt709", "-color_primaries", "bt709", "-color_trc", "bt709", "-color_range", "tv", | |
| "-c:a", "aac", "-profile:a", "aac_low", "-ar", "48000", "-ac", "2", "-b:a", "192k", | |
| "-af", "aresample=async=1:min_hard_comp=0.100000:first_pts=0", | |
| "-movflags", "+faststart", | |
| "-fflags", "+genpts", "-max_interleave_delta", "100M", | |
| "-vsync", "cfr", "-fps_mode", "cfr", | |
| "-shortest", | |
| str(final_output) | |
| ] | |
| res = subprocess.run(cmd_render, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True) | |
| if res.returncode != 0: | |
| self._log(f"⚠️ Fallback render không mask: {res.stderr[:120]} | retry with minimal TikTok spec") | |
| # Minimal fallback: at least ensure TikTok audio/video spec | |
| fallback_cmd = [ | |
| str(self.ffmpeg_path), "-y", | |
| "-fflags", "+genpts", | |
| "-i", str(v_path), | |
| "-i", str(mixed_audio), | |
| "-vf", f"{tiktok_polish},{sub_filter}", | |
| "-map", "0:v:0", "-map", "1:a:0", | |
| "-r", "30", | |
| "-c:v", "libx264", "-preset", "medium", "-crf", "18", | |
| "-profile:v", "high", "-pix_fmt", "yuv420p", | |
| "-c:a", "aac", "-profile:a", "aac_low", "-ar", "48000", "-ac", "2", "-b:a", "192k", | |
| "-movflags", "+faststart", | |
| "-shortest", | |
| str(final_output) | |
| ] | |
| subprocess.run(fallback_cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True) | |
| total_elapsed = time.time() - start_time | |
| # Verify output duration vs input — báo động ngay nếu bị cắt ngọn (2p->20s) | |
| try: | |
| out_dur = self._get_video_duration_sec(final_output) | |
| in_dur = video_dur_sec or self._get_video_duration_sec(v_path) | |
| if in_dur > 0 and out_dur > 0: | |
| if out_dur < in_dur - 3.0: | |
| self._log(f"❌ CẢNH BÁO ĐỘ DÀI: input {in_dur:.1f}s nhưng output chỉ {out_dur:.1f}s " | |
| f"(thiếu {in_dur - out_dur:.1f}s)! Kiểm tra phụ đề/ASR-OCR có bị rụng đuôi không.") | |
| else: | |
| self._log(f"⏱️ Verify độ dài OK: input {in_dur:.1f}s -> output {out_dur:.1f}s.") | |
| except Exception: | |
| pass | |
| self.progress_fn(100, "DONE") | |
| self._log(f"🎉 HOÀN THÀNH XUẤT SẮC TRONG {total_elapsed:.1f}s!") | |
| self._log(f"📁 Video đã lưu tại: {final_output}") | |
| return str(final_output) | |
| def run_video( | |
| self, | |
| video_path: str, | |
| mode: str = "asr", | |
| source_lang: str = "zh", | |
| voice: str = "vi-VN-NamMinhNeural", | |
| speed: float = 1.0, | |
| pitch: int = 0, | |
| volume: int = 100, | |
| region_preset: str = "custom", | |
| sub_mask_mode: str = "box", | |
| sub_style: str = "motion_drip", | |
| custom_y_pct: float = 75.0, | |
| custom_h_pct: float = 20.0, | |
| custom_x_pct: float = 0.0, | |
| custom_w_pct: float = 100.0, | |
| ducking_ratio: float = 0.18, | |
| enable_sfx: bool = True, | |
| mute_original_audio: bool = False, | |
| logo_enabled: bool = False, | |
| logo_path: str = "", | |
| logo_preset: str = "bottom_right", | |
| logo_scale: float = 15.0, | |
| logo_opacity: float = 1.0, | |
| logo_x_pct: float = 80.0, | |
| logo_y_pct: float = 80.0, | |
| logo_chromakey: bool = True, | |
| cta_enabled: bool = False, | |
| cta_path: str = "", | |
| cta_preset: str = "bottom_center", | |
| cta_scale: float = 35.0, | |
| cta_opacity: float = 1.0, | |
| cta_x_pct: float = 50.0, | |
| cta_y_pct: float = 85.0, | |
| cta_interval: float = 5.0, | |
| cta_duration: float = 2.0, | |
| cta_chromakey: bool = True | |
| ) -> Optional[str]: | |
| """Full 1-Click End-to-End Automatic Dubbing Pipeline.""" | |
| v_path = Path(video_path) | |
| if not v_path.exists(): | |
| self._log(f"❌ Video not found: {video_path}") | |
| return None | |
| # Register into SQLite Job DB | |
| try: | |
| self.job_manager.register_job(str(v_path)) | |
| except Exception: | |
| pass | |
| try: | |
| # 1. Extract subtitles | |
| subs_res = self.extract_subtitles_only( | |
| video_path=video_path, | |
| mode=mode, | |
| source_lang=source_lang, | |
| region_preset=region_preset, | |
| custom_y_pct=custom_y_pct, | |
| custom_h_pct=custom_h_pct, | |
| custom_x_pct=custom_x_pct, | |
| custom_w_pct=custom_w_pct | |
| ) | |
| if not subs_res: | |
| return None | |
| # 2. Render from subtitles | |
| return self.render_from_subtitles( | |
| video_path=video_path, | |
| subtitles_data=subs_res["blocks"], | |
| voice=voice, | |
| speed=speed, | |
| pitch=pitch, | |
| volume=volume, | |
| region_preset=region_preset, | |
| sub_mask_mode=sub_mask_mode, | |
| sub_style=sub_style, | |
| custom_y_pct=custom_y_pct, | |
| custom_h_pct=custom_h_pct, | |
| custom_x_pct=custom_x_pct, | |
| custom_w_pct=custom_w_pct, | |
| ducking_ratio=ducking_ratio, | |
| enable_sfx=enable_sfx, | |
| mute_original_audio=mute_original_audio, | |
| logo_enabled=logo_enabled, | |
| logo_path=logo_path, | |
| logo_preset=logo_preset, | |
| logo_scale=logo_scale, | |
| logo_opacity=logo_opacity, | |
| logo_x_pct=logo_x_pct, | |
| logo_y_pct=logo_y_pct, | |
| logo_chromakey=logo_chromakey, | |
| cta_enabled=cta_enabled, | |
| cta_path=cta_path, | |
| cta_preset=cta_preset, | |
| cta_scale=cta_scale, | |
| cta_opacity=cta_opacity, | |
| cta_x_pct=cta_x_pct, | |
| cta_y_pct=cta_y_pct, | |
| cta_interval=cta_interval, | |
| cta_duration=cta_duration, | |
| cta_chromakey=cta_chromakey | |
| ) | |
| except Exception as e: | |
| self._log(f"❌ LỖI PIPELINE: {str(e)}") | |
| self.progress_fn(0, "FAILED") | |
| return None | |
| def _convert_srt_to_ass( | |
| self, | |
| srt_path: Path, | |
| ass_path: Path, | |
| vw: int, | |
| vh: int, | |
| box_x: int, | |
| box_y: int, | |
| box_w: int, | |
| box_h: int, | |
| sub_style: str = "motion_drip" | |
| ): | |
| """Converts SRT to styled ASS format for TikTok/Shorts typography.""" | |
| content = srt_path.read_text(encoding="utf-8", errors="ignore") | |
| # TikTok safe zone: keep subtitle inside 82% height, 7% bottom margin, larger font for mobile legibility | |
| safe_margin_v = max(42, int(vh * 0.07)) | |
| raw_margin_v = int(vh - (box_y + box_h * 0.75)) if box_h > 0 else safe_margin_v | |
| margin_v = max(safe_margin_v, raw_margin_v) | |
| # Clamp to avoid bottom UI overlap | |
| max_bottom = int(vh * 0.12) | |
| if margin_v < max_bottom: | |
| margin_v = max_bottom | |
| # Larger font for TikTok: vertical 56-72, horizontal 48-64 | |
| if box_h > 0: | |
| base = int(box_h * 0.52) | |
| if vh > vw: # vertical TikTok | |
| font_size = max(42, min(72, base)) | |
| else: | |
| font_size = max(38, min(64, base)) | |
| else: | |
| font_size = 58 if vh > vw else 50 | |
| # Typography Color & Outline palettes - TikTok optimized for high compression | |
| if sub_style == "motion_drip" or sub_style == "yellow_bold": | |
| font_color = "&H0000FFFF" # Bright Yellow | |
| outline_color = "&H00000000" # Pure Black | |
| back_color = "&H99000000" | |
| outline = 5.0 | |
| shadow = 2.2 | |
| elif sub_style == "neon_cyan": | |
| font_color = "&H00FFFF00" # Cyan | |
| outline_color = "&H00000000" | |
| back_color = "&H99000000" | |
| outline = 5.0 | |
| shadow = 2.2 | |
| elif sub_style == "capsule_tag": | |
| font_color = "&H00FFFFFF" # White | |
| outline_color = "&H00111111" | |
| back_color = "&HBB000000" | |
| outline = 3.2 | |
| shadow = 0.0 | |
| else: # white_bold | |
| font_color = "&H00FFFFFF" # Crisp White | |
| outline_color = "&H00000000" | |
| back_color = "&H99000000" | |
| outline = 4.8 | |
| shadow = 2.0 | |
| ass_header = f"""[Script Info] | |
| ScriptType: v4.00+ | |
| PlayResX: {vw} | |
| PlayResY: {vh} | |
| ScaledBorderAndShadow: yes | |
| [V4+ Styles] | |
| Format: Name, Fontname, Fontsize, PrimaryColour, SecondaryColour, OutlineColour, BackColour, Bold, Italic, Underline, StrikeOut, ScaleX, ScaleY, Spacing, Angle, BorderStyle, Outline, Shadow, Alignment, MarginL, MarginR, MarginV, Encoding | |
| Style: Default,DejaVu Sans,{font_size},{font_color},&H000000FF,{outline_color},{back_color},1,0,0,0,100,100,0,0,1,{outline},{shadow},2,40,40,{margin_v},1 | |
| [Events] | |
| Format: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text | |
| """ | |
| events = [] | |
| pattern = r"(\d+)\s+(\d{2}:\d{2}:\d{2}[.,]\d{3})\s*-->\s*(\d{2}:\d{2}:\d{2}[.,]\d{3})\s*\n(.*?)(?=\n\s*\d+\s+\d{2}:\d{2}:\d{2}[.,]\d{3}\s*-->|\Z)" | |
| for m in re.finditer(pattern, content, re.DOTALL): | |
| start_str = m.group(2).strip().replace(",", ".") | |
| end_str = m.group(3).strip().replace(",", ".") | |
| s_parts = start_str.split(":") | |
| e_parts = end_str.split(":") | |
| s_ass = f"{int(s_parts[0])}:{s_parts[1]}:{float(s_parts[2]):05.2f}" | |
| e_ass = f"{int(e_parts[0])}:{e_parts[1]}:{float(e_parts[2]):05.2f}" | |
| text = " ".join(line.strip() for line in m.group(4).splitlines() if line.strip()) | |
| if text: | |
| # FIX: escape ký tự ASS đặc biệt — "{deal}", "\", emoticon bị libass | |
| # hiểu thành override-tag -> render lỗi/hiện chữ thô | |
| text = text.replace("\\", "\\\\").replace("{", "\\{").replace("}", "\\}") | |
| events.append(f"Dialogue: 0,{s_ass},{e_ass},Default,,0,0,0,,{text}") | |
| ass_path.write_text(ass_header + "\n".join(events), encoding="utf-8") | |
| def _parse_srt_blocks(self, content: str) -> List[Dict]: | |
| # FIX: giờ 1 chữ số (\d{1,2}) — timestamp "0:00:01,000" của Whisper/tool lẻ bị miss cả block | |
| pattern = r"(\d+)\s+(\d{1,2}:\d{2}:\d{2}[.,]\d{3})\s*-->\s*(\d{1,2}:\d{2}:\d{2}[.,]\d{3})\s*\n(.*?)(?=\n\s*\d+\s+\d{1,2}:\d{2}:\d{2}[.,]\d{3}\s*-->|\Z)" | |
| blocks = [] | |
| for m in re.finditer(pattern, content, re.DOTALL): | |
| text = " ".join(line.strip() for line in m.group(4).splitlines() if line.strip()) | |
| start_str = m.group(2).strip() | |
| end_str = m.group(3).strip() | |
| if text: | |
| blocks.append({ | |
| "id": int(m.group(1)), | |
| "timing": f"{start_str} --> {end_str}", | |
| "text": text, | |
| "start_ms": self._ts_to_ms(start_str), | |
| "end_ms": self._ts_to_ms(end_str), | |
| }) | |
| return blocks | |
| def _ts_to_ms(ts: str) -> int: | |
| ts = ts.strip().replace(".", ",") | |
| m = re.match(r"(\d+):(\d+):(\d+)[,](\d+)", ts) | |
| if m: | |
| h, mins, s, ms = map(int, m.groups()) | |
| return ((h * 3600 + mins * 60 + s) * 1000) + ms | |
| return 0 | |
| def _build_system_prompt(self) -> str: | |
| glossary_sample = ", ".join([f"{k} -> {v}" for k, v in list(self.glossary.items())[:35]]) | |
| return ( | |
| "Bạn là chuyên gia dịch thuật video Sneaker, Thời trang Streetwear, Review sản phẩm từ tiếng Trung/Anh sang tiếng Việt tự nhiên, sành điệu, bắt trend Gen Z.\n" | |
| "QUY TẮC BẮT BUỘC:\n" | |
| "1. Dịch từng dòng theo cấu trúc: [N] Câu dịch tiếng Việt hoàn chỉnh.\n" | |
| "2. Giữ nguyên thuật ngữ & thương hiệu tiếng Anh (Nike, Jordan, Yeezy, BAPE, Supreme, Rick Owens, outfit, fit, drip, full box, collab, signature...). \n" | |
| f"3. Bắt buộc áp dụng từ điển chuyên ngành: {glossary_sample}\n" | |
| "4. Dịch ĐẦY ĐỦ Ý NGHĨA trọn vẹn của câu, giữ đủ các từ khoá, TUYỆT ĐỐI KHÔNG cắt xén, không bỏ lửng hay cắt cụt mất từ ở đầu, giữa hoặc cuối câu. Câu dịch phải trọn vẹn ngữ nghĩa, tự nhiên và dễ hiểu.\n" | |
| "5. KHÔNG giải thích, CHỈ trả về danh sách các dòng [N] Tiếng Việt." | |
| ) | |
| def _translate_srt_cloud(self, srt_in: Path, srt_out: Path, source_lang: str = "zh"): | |
| content = srt_in.read_text(encoding="utf-8", errors="ignore") | |
| blocks = self._parse_srt_blocks(content) | |
| if not blocks: | |
| srt_out.write_text(content, encoding="utf-8") | |
| return | |
| system_prompt = self._build_system_prompt() | |
| validator = TranslationValidator() | |
| trans_map = {} | |
| # Phase 1: Batch translation in chunks of 20 | |
| chunk_size = 20 | |
| for i in range(0, len(blocks), chunk_size): | |
| chunk = blocks[i : i + chunk_size] | |
| res = self._translate_chunk_with_retry(chunk, system_prompt, validator, max_retries=2) | |
| trans_map.update(res) | |
| # Phase 2: Retry missing / Chinese-leaked blocks | |
| missing_blocks = [b for b in blocks if (b["id"] not in trans_map or has_chinese(trans_map.get(b["id"], "")))] | |
| if missing_blocks: | |
| self._log(f"🔄 Đang hoàn thiện nốt {len(missing_blocks)} câu dịch còn lại...") | |
| retry_result = self._translate_chunk_with_retry(missing_blocks, system_prompt, validator, max_retries=2) | |
| trans_map.update(retry_result) | |
| # Phase 3: Google Translate fallback | |
| still_missing = [b for b in blocks if (b["id"] not in trans_map or has_chinese(trans_map.get(b["id"], "")))] | |
| if still_missing: | |
| self._log(f"🌐 {len(still_missing)} câu cần fallback Google Translate...") | |
| google_fb = GoogleTranslateFallback(log_fn=self._log) | |
| google_result = google_fb.translate_blocks(still_missing) | |
| for bid_str, text in google_result.items(): | |
| trans_map[int(bid_str)] = text | |
| # Phase 4: Post-editing | |
| self._log("✨ Chạy Post-Editor sửa lỗi dịch...") | |
| for b in blocks: | |
| bid = b["id"] | |
| if bid in trans_map: | |
| trans_map[bid] = post_edit_translation(b["text"], trans_map[bid]) | |
| # Phase 5: Quality Guard | |
| self._log("🛡️ Quality Guard kiểm tra chất lượng bản dịch...") | |
| guard = TranslationQualityGuard(min_score=70) | |
| guard_dict = {str(b["id"]): trans_map.get(b["id"], b["text"]) for b in blocks} | |
| fixed_dict, quality_report = guard.audit_and_fix(blocks, guard_dict) | |
| for b in blocks: | |
| bid_str = str(b["id"]) | |
| if bid_str in fixed_dict: | |
| trans_map[b["id"]] = fixed_dict[bid_str] | |
| # Phase 6: Subtitle text cleanup (safe formatting) | |
| for b in blocks: | |
| bid = b["id"] | |
| if bid in trans_map: | |
| t = str(trans_map[bid]).strip() | |
| t = re.sub(r"\s+", " ", t).strip(" ,") | |
| trans_map[bid] = t | |
| # Phase 7: Output SRT | |
| out_lines = [] | |
| for b in blocks: | |
| vi_text = trans_map.get(b["id"], b["text"]) | |
| out_lines.append(str(b["id"])) | |
| out_lines.append(b["timing"]) | |
| out_lines.append(vi_text) | |
| out_lines.append("") | |
| srt_out.write_text("\n".join(out_lines), encoding="utf-8") | |
| self._log(f"✅ Dịch hoàn tất {len(blocks)} câu với 7 bước kiểm tra chất lượng.") | |
| def _translate_chunk_with_retry( | |
| self, | |
| chunk: List[Dict], | |
| system_prompt: str, | |
| validator: TranslationValidator, | |
| max_retries: int = 2 | |
| ) -> Dict[int, str]: | |
| prompt_lines = [f"[{b['id']}] {b['text']}" for b in chunk] | |
| full_transcript = "\n".join(prompt_lines) | |
| result = {} | |
| for attempt in range(max_retries + 1): | |
| raw_res = self._direct_25_model_translate(system_prompt, full_transcript) | |
| cleaned_res = re.sub(r"<think>.*?</think>", "", raw_res, flags=re.DOTALL).strip() | |
| attempt_map = {} | |
| for line in cleaned_res.splitlines(): | |
| m = re.match(r"^\s*\[(\d+)\]\s*(.*)$", line.strip()) | |
| if m: | |
| attempt_map[int(m.group(1))] = m.group(2).strip() | |
| # FIX rụng đuôi dịch: chỉ accept early-return khi ĐỦ 100% số dòng chunk. | |
| # Trước đây subset 5/20 dòng pass validate -> return thiếu 15 dòng im lặng. | |
| expected_ids = {b["id"] for b in chunk} | |
| got_ids = set(attempt_map.keys()) & expected_ids | |
| val_dict = {str(b["id"]): attempt_map.get(b["id"], "") for b in chunk if b["id"] in attempt_map} | |
| val_chunk = [b for b in chunk if b["id"] in attempt_map] | |
| if val_chunk and val_dict: | |
| is_valid, error_msg = validator.validate(val_chunk, val_dict) | |
| if is_valid and got_ids == expected_ids: | |
| result.update(attempt_map) | |
| return result | |
| if not is_valid: | |
| self._log(f"⚠️ Chunk validate fail ({len(got_ids)}/{len(expected_ids)} dòng): {error_msg} — retry...") | |
| elif got_ids != expected_ids: | |
| missing = sorted(expected_ids - got_ids) | |
| self._log(f"⚠️ Chunk thiếu {len(missing)}/{len(expected_ids)} dòng {missing[:8]} — retry...") | |
| for bid, text in attempt_map.items(): | |
| src_block = next((b for b in chunk if b["id"] == bid), None) | |
| if src_block: | |
| ok, _ = validator.validate_single_block(bid, src_block["text"], text) | |
| if ok: | |
| result[bid] = text | |
| elif attempt_map: | |
| # Không parse được dòng nào theo format [N] — giữ lại để Phase 2/3 xử lý | |
| self._log(f"⚠️ Chunk parse được 0/{len(expected_ids)} dòng hợp lệ — retry...") | |
| return result | |
| NIM_TRANSLATION_MODEL = "nvidia/nemotron-3.5-lightning-30b-a3b" | |
| NIM_TRANSLATION_URL = "https://integrate.api.nvidia.com/v1/chat/completions" | |
| def _get_nim_keys(self): | |
| """Lấy Nvidia NIM keys từ .env/env: NIM_API_KEY > NVIDIA_KEY_1..N.""" | |
| import os | |
| keys = [] | |
| direct = (os.getenv("NIM_API_KEY", "") or "").strip() | |
| if direct: | |
| keys.append(direct) | |
| # Ưu tiên engine đã load sẵn (.env + os.environ) | |
| for src in (getattr(self.asr_engine, "nvidia_keys", []), | |
| getattr(self.ocr_engine, "nvidia_keys", [])): | |
| for k in (src or []): | |
| if k and k not in keys: | |
| keys.append(k) | |
| # Quét NVIDIA_KEY_N trực tiếp từ environ (cho HF Secrets) | |
| for i in range(1, 20): | |
| v = (os.getenv(f"NVIDIA_KEY_{i}", "") or "").strip() | |
| if v and v not in keys: | |
| keys.append(v) | |
| return keys | |
| def _direct_25_model_translate(self, system_prompt: str, user_content: str) -> str: | |
| messages = [ | |
| {"role": "system", "content": system_prompt}, | |
| {"role": "user", "content": user_content} | |
| ] | |
| # 0. NVIDIA NIM Nemotron 3.5 Lightning 30B (ƯU TIÊN #1 — key có sẵn trong tool) | |
| nim_keys = self._get_nim_keys() | |
| for key in nim_keys: | |
| try: | |
| res = requests.post( | |
| self.NIM_TRANSLATION_URL, | |
| headers={"Authorization": f"Bearer {key}", "Content-Type": "application/json"}, | |
| json={"model": self.NIM_TRANSLATION_MODEL, "messages": messages, | |
| "temperature": 0.2, "max_tokens": 4096}, | |
| timeout=30 | |
| ) | |
| if res.status_code == 200: | |
| content = res.json()["choices"][0]["message"]["content"].strip() | |
| content = re.sub(r"<think>.*?</think>", "", content, flags=re.DOTALL).strip() | |
| if len(content) > 15: | |
| self._log(f"✅ Dịch bằng NVIDIA NIM ({self.NIM_TRANSLATION_MODEL})") | |
| return content | |
| else: | |
| self._log(f"⚠️ NIM {res.status_code}: {res.text[:120]}") | |
| except Exception as e: | |
| self._log(f"⚠️ NIM lỗi: {e}") | |
| continue | |
| # 1. Google Gemini (Fastest & highest accuracy for Asian languages) | |
| gemini_keys = self.ocr_engine.gemini_keys | |
| for model in ["gemini-3.6-flash", "gemini-3.5-flash", "gemini-flash-latest", "gemini-pro-latest", "gemini-2.5-flash", "gemini-2.0-flash"]: | |
| for key in gemini_keys: | |
| try: | |
| url = f"https://generativelanguage.googleapis.com/v1beta/models/{model}:generateContent?key={key}" | |
| payload = { | |
| "contents": [{"parts": [{"text": f"{system_prompt}\n\n{user_content}"}]}], | |
| "generationConfig": {"temperature": 0.2, "maxOutputTokens": 4096} | |
| } | |
| res = requests.post(url, json=payload, timeout=25) | |
| if res.status_code == 200: | |
| content = res.json()["candidates"][0]["content"]["parts"][0]["text"].strip() | |
| if len(content) > 15: | |
| return content | |
| except Exception: | |
| pass | |
| # 2. Groq (High speed LLM) | |
| groq_keys = self.asr_engine.groq_keys | |
| for model in ["openai/gpt-oss-20b", "groq/compound", "qwen/qwen3.6-27b", "llama-3.3-70b-versatile", "llama-3.1-8b-instant"]: | |
| for key in groq_keys: | |
| try: | |
| res = requests.post( | |
| "https://api.groq.com/openai/v1/chat/completions", | |
| headers={"Authorization": f"Bearer {key}", "Content-Type": "application/json"}, | |
| json={"model": model, "messages": messages, "temperature": 0.2, "max_tokens": 4096}, | |
| timeout=25 | |
| ) | |
| if res.status_code == 200: | |
| content = res.json()["choices"][0]["message"]["content"].strip() | |
| content = re.sub(r"<think>.*?</think>", "", content, flags=re.DOTALL).strip() | |
| if len(content) > 15: | |
| return content | |
| except Exception: | |
| pass | |
| # 3. OpenRouter Free & Standard Models | |
| or_keys = self.ocr_engine.openrouter_keys | |
| for model in [ | |
| "nvidia/nemotron-3.5-lightning:free", | |
| "inclusionai/ling-3.0-flash-fin:free", | |
| "liquid/lfm-2.5-2.6b:free", | |
| "deepseek/deepseek-r1:free", | |
| "qwen/qwen-2.5-72b-instruct:free" | |
| ]: | |
| for key in or_keys: | |
| try: | |
| res = requests.post( | |
| "https://openrouter.ai/api/v1/chat/completions", | |
| headers={"Authorization": f"Bearer {key}", "Content-Type": "application/json", "HTTP-Referer": "https://trungsangviet.local", "X-Title": "TrungSangViet"}, | |
| json={"model": model, "messages": messages, "temperature": 0.2, "max_tokens": 4096}, | |
| timeout=30 | |
| ) | |
| if res.status_code == 200: | |
| content = res.json()["choices"][0]["message"]["content"].strip() | |
| content = re.sub(r"<think>.*?</think>", "", content, flags=re.DOTALL).strip() | |
| if len(content) > 15: | |
| return content | |
| except Exception: | |
| pass | |
| # FIX: trước đây fail hết model thì trả source (tiếng Trung) như "đã dịch" -> | |
| # caller parse thành trans_map, có thể lọt qua validator vào SRT. | |
| # Nay trả "" để Phase 2 retry + Phase 3 Google fallback làm việc đúng. | |
| self._log("❌ Tất cả translation models đều fail cho chunk này — để trống cho fallback.") | |
| return "" | |