DRIPPY4 / app /core /cloud_pipeline.py
hoangtaiii's picture
Fix 2p->20s + 20 bugs: pad audio, OCR/ASR coverage gates, TTS cache/placeholder, omni cloud, ASS escape, timeout, preflight, pool locks, SSRF guards
df03342 verified
Raw History Blame Contribute Delete
66.7 kB
"""
app/core/cloud_pipeline.py
──────────────────────────
Pure Cloud-Native Headless Pipeline Coordinator.
Features:
1. High-Definition Video Preview Frame Generator using FFmpeg (100% Reliable).
2. Multi-Model ASR Matrix (Groq Whisper Large V3 / OpenRouter / Gemini).
3. AI Vision Models OCR with automatic ASR fallback.
4. Active 2026 AI Translation Models (Groq Qwen 3.6 / GPT-OSS 120B / Gemini Flash / Nemotron 3).
5. Glossary & Slang Mapping (Fashion, Sneaker, Streetwear, Chinese Slang).
6. 7-Step Quality Guard & Semantic Validation & Timing Compaction.
7. Subtitle Editing Hooks (extract_subtitles_only & render_from_subtitles).
8. ASS Subtitle Engine with TikTok / Shorts Typography & Word Jump Styles.
9. 100% Solid/Delogo Masking Box to eliminate old Chinese subtitles.
10. Studio SFX & Audio Ducking Mixer.
11. SQLite JobManager Integration for state persistence.
"""
import os
import re
import sys
import json
import time
import base64
import shutil
import cv2
import requests
import subprocess
from pathlib import Path
from typing import Optional, Callable, Dict, Any, Tuple, List
from app.core.cloud_asr import CloudASREngine
from app.core.cloud_ocr import CloudOCREngine
from app.core.cloud_tts import CloudTTSEngine
from app.core.vietnamese_text_normalizer import VietnameseTextNormalizer
from app.core.translation_validator import TranslationValidator
from app.core.translation_quality_guard import TranslationQualityGuard
from app.core.translation_post_editor import post_edit_translation
from app.core.google_translate_fallback import GoogleTranslateFallback
from app.core.subtitle_compactor import compact_blocks_for_tts
from app.core.job_manager import JobManager
from app.core.studio_sfx import generate_sfx_clip
def has_chinese(text: str) -> bool:
"""Returns True if string contains CJK Chinese characters."""
return bool(re.search(r"[\u4e00-\u9fff]", text))
class CloudPipeline:
def __init__(
self,
base_dir: Optional[Path] = None,
log_callback: Optional[Callable[[str], None]] = None,
progress_callback: Optional[Callable[[int, str], None]] = None,
ffmpeg_path: str = "ffmpeg"
):
self.base_dir = Path(base_dir or Path(__file__).resolve().parents[2])
self.output_dir = self.base_dir / "output"
self.temp_dir = self.base_dir / "temp"
self.output_dir.mkdir(parents=True, exist_ok=True)
self.temp_dir.mkdir(parents=True, exist_ok=True)
self.log_fn = log_callback or print
self.progress_fn = progress_callback or (lambda pct, stage: None)
self.ffmpeg_path = ffmpeg_path
self.asr_engine = CloudASREngine(log_fn=self._log)
self.ocr_engine = CloudOCREngine(log_fn=self._log)
self.tts_engine = CloudTTSEngine(log_fn=self._log, ffmpeg_path=self.ffmpeg_path)
self.normalizer = VietnameseTextNormalizer()
self.job_manager = JobManager.instance()
self.glossary = self._load_glossary()
def _log(self, msg: str):
self.log_fn(f"[Cloud Pipeline] {msg}")
def _get_video_duration_sec(self, v_path: Path) -> float:
"""Probe video duration (sec). ffprobe -> cv2 fallback. 0.0 if unknown."""
try:
import shutil as _sh
_ffprobe = _sh.which("ffprobe")
if not _ffprobe:
cand = Path(self.ffmpeg_path).parent / "ffprobe.exe"
_ffprobe = str(cand) if cand.exists() else "ffprobe"
_res = subprocess.run(
[_ffprobe, "-v", "error", "-show_entries", "format=duration",
"-of", "default=noprint_wrappers=1:nokey=1", str(v_path)],
capture_output=True, text=True, timeout=15)
_dur = float((_res.stdout or "").strip())
if _dur > 0:
return _dur
except Exception:
pass
try:
cap = cv2.VideoCapture(str(v_path))
fps = cap.get(cv2.CAP_PROP_FPS) or 0
frames = cap.get(cv2.CAP_PROP_FRAME_COUNT) or 0
cap.release()
if fps > 0 and frames > 0:
return float(frames / fps)
except Exception:
pass
return 0.0
def _pad_audio_to_video_duration(self, audio_path: Path, video_dur: float) -> bool:
"""Pad (never cut) audio to exactly video duration. Guards against
`-shortest` slicing the video when dubbing/mix is shorter than source
(e.g. mute_original + truncated subtitles: 2min video -> 20s output)."""
if video_dur <= 0 or not audio_path.exists():
return False
try:
tmp = audio_path.parent / f"{audio_path.stem}_padded.wav"
cmd = [str(self.ffmpeg_path), "-y", "-i", str(audio_path),
"-filter:a", f"apad=whole_dur={video_dur:.3f}",
"-ac", "2", "-ar", "48000", "-t", f"{video_dur:.3f}", str(tmp)]
res = subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.PIPE, timeout=120)
if res.returncode == 0 and tmp.exists() and tmp.stat().st_size > 1000:
tmp.replace(audio_path)
return True
self._log(f"⚠️ Pad audio to {video_dur:.1f}s failed, keeping original mix.")
try:
if tmp.exists():
tmp.unlink()
except Exception:
pass
except Exception as e:
self._log(f"⚠️ Pad audio error: {e}")
return False
def _load_glossary(self) -> Dict[str, str]:
glossary_path = self.base_dir / "glossary.json"
if glossary_path.exists():
try:
data = json.loads(glossary_path.read_text(encoding="utf-8"))
return data.get("glossary", {})
except Exception as e:
self._log(f"⚠️ Lỗi đọc glossary.json: {e}")
return {}
def calculate_region(
self,
video_path: Path,
region_preset: str = "custom",
custom_y_pct: float = 75.0,
custom_h_pct: float = 20.0,
custom_x_pct: float = 0.0,
custom_w_pct: float = 100.0
) -> Tuple[int, int, int, int]:
vw, vh = 1920, 1080
try:
cap = cv2.VideoCapture(str(video_path))
w = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH))
h = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT))
cap.release()
if w > 0 and h > 0:
vw, vh = w, h
except Exception:
pass
if region_preset == "bottom_25":
x = 0
w = vw
y = int(vh * 0.72)
h = int(vh * 0.24)
elif region_preset == "bottom_15":
x = 0
w = vw
y = int(vh * 0.82)
h = int(vh * 0.15)
elif region_preset == "middle_25":
x = 0
w = vw
y = int(vh * 0.38)
h = int(vh * 0.24)
elif region_preset == "none":
return (0, 0, 0, 0)
elif region_preset == "custom":
x = int(vw * (custom_x_pct / 100.0))
w = int(vw * (custom_w_pct / 100.0))
y = int(vh * (custom_y_pct / 100.0))
h = int(vh * (custom_h_pct / 100.0))
else:
return (0, 0, 0, 0)
x = max(0, min(vw - 2, x))
y = max(0, min(vh - 2, y))
w = max(2, min(vw - x, w))
h = max(2, min(vh - y, h))
return (x, y, w, h)
def generate_preview_frame(
self,
video_path: str,
region_preset: str = "custom",
custom_y_pct: float = 75.0,
custom_h_pct: float = 20.0,
custom_x_pct: float = 0.0,
custom_w_pct: float = 100.0,
sec: float = 2.0,
return_type: str = "rgb",
# ── Logo overlay params ──
logo_enabled: bool = False,
logo_path: str = "",
logo_preset: str = "bottom_right",
logo_scale: float = 15.0,
logo_opacity: float = 1.0,
logo_x_pct: float = 80.0,
logo_y_pct: float = 80.0,
logo_chromakey: bool = True,
# ── CTA video overlay params (appears every 5s) ──
cta_enabled: bool = False,
cta_path: str = "",
cta_preset: str = "bottom_center",
cta_scale: float = 35.0,
cta_opacity: float = 1.0,
cta_x_pct: float = 50.0,
cta_y_pct: float = 85.0,
cta_interval: float = 5.0,
cta_duration: float = 2.0,
cta_chromakey: bool = True
):
"""Extracts a frame at timestamp sec and overlays bounding box + logo + CTA preview."""
v_file = Path(video_path)
if not v_file.exists():
return None
raw_frame_path = self.temp_dir / f"raw_frame_{int(time.time()*1000)}.jpg"
cmd = [
str(self.ffmpeg_path), "-ss", str(sec), "-y",
"-i", str(v_file),
"-frames:v", "1", "-q:v", "2",
str(raw_frame_path)
]
try:
subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True)
frame = cv2.imread(str(raw_frame_path))
if raw_frame_path.exists():
raw_frame_path.unlink()
except Exception:
frame = None
if frame is None:
cap = cv2.VideoCapture(str(v_file))
fps = cap.get(cv2.CAP_PROP_FPS) or 25.0
target_frame = max(0, int(sec * fps))
cap.set(cv2.CAP_PROP_POS_FRAMES, target_frame)
ret, frame = cap.read()
cap.release()
if frame is None:
return None
x, y, w, h = self.calculate_region(
v_file, region_preset, custom_y_pct, custom_h_pct, custom_x_pct, custom_w_pct
)
if w > 0 and h > 0:
overlay = frame.copy()
cv2.rectangle(overlay, (x, y), (x + w, y + h), (0, 0, 255), -1)
cv2.addWeighted(overlay, 0.35, frame, 0.65, 0, frame)
cv2.rectangle(frame, (x, y), (x + w, y + h), (0, 255, 255), 3)
label = "VUNG CHE SUB CU & QUET OCR"
font = cv2.FONT_HERSHEY_SIMPLEX
font_scale = max(0.5, frame.shape[1] / 1200.0)
thickness = 2
(tw, th), _ = cv2.getTextSize(label, font, font_scale, thickness)
label_y = max(th + 10, y - 8)
cv2.rectangle(frame, (x, label_y - th - 6), (x + tw + 10, label_y + 4), (0, 0, 0), -1)
cv2.putText(frame, label, (x + 5, label_y), font, font_scale, (0, 255, 255), thickness, cv2.LINE_AA)
# ── Logo overlay preview ──
if logo_enabled and logo_path and Path(logo_path).exists():
try:
from app.core.logo_overlay import load_logo_rgba, overlay_logo_on_frame
_logo_rgba = load_logo_rgba(logo_path, chromakey=logo_chromakey)
if _logo_rgba is not None:
frame = overlay_logo_on_frame(
frame_bgr=frame,
logo_rgba=_logo_rgba,
preset=logo_preset,
scale_pct=float(logo_scale),
opacity=float(logo_opacity),
custom_x_pct=float(logo_x_pct),
custom_y_pct=float(logo_y_pct),
margin_pct=2.0
)
cv2.putText(frame, f"LOGO:{logo_preset} {logo_scale:.0f}%", (10, 30), cv2.FONT_HERSHEY_SIMPLEX, 0.7, (0, 255, 255), 2, cv2.LINE_AA)
except Exception as _e:
self._log(f"⚠️ Logo preview error: {_e}")
# ── CTA video overlay preview (every 5s) ──
if cta_enabled and cta_path and Path(cta_path).exists():
try:
from app.core.logo_overlay import load_cta_frame_rgba, overlay_cta_on_frame, should_show_cta_at_time
if should_show_cta_at_time(float(sec), float(cta_interval), float(cta_duration)):
# CTA time loops within CTA video duration
_cta_time = float(sec) % float(cta_interval)
# If CTA video is shorter than interval, loop inside
_cta_rgba = load_cta_frame_rgba(cta_path, cta_time_sec=_cta_time, chromakey=cta_chromakey)
if _cta_rgba is not None:
frame = overlay_cta_on_frame(
frame_bgr=frame,
cta_rgba=_cta_rgba,
preset=cta_preset,
scale_pct=float(cta_scale),
opacity=float(cta_opacity),
custom_x_pct=float(cta_x_pct),
custom_y_pct=float(cta_y_pct),
margin_pct=2.0
)
cv2.putText(frame, f"CTA:{cta_preset} {cta_scale:.0f}% every {cta_interval:.0f}s", (10, 60), cv2.FONT_HERSHEY_SIMPLEX, 0.6, (0, 255, 0), 2, cv2.LINE_AA)
else:
cv2.putText(frame, f"CTA: hidden at {sec:.1f}s (interval {cta_interval:.0f}s)", (10, 60), cv2.FONT_HERSHEY_SIMPLEX, 0.5, (0, 165, 255), 1, cv2.LINE_AA)
except Exception as _e:
self._log(f"⚠️ CTA preview error: {_e}")
if return_type == "base64":
_, buffer = cv2.imencode('.jpg', frame, [cv2.IMWRITE_JPEG_QUALITY, 85])
return base64.b64encode(buffer).decode('utf-8')
return cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)
def extract_subtitles_only(
self,
video_path: str,
mode: str = "asr",
source_lang: str = "zh",
region_preset: str = "custom",
custom_y_pct: float = 75.0,
custom_h_pct: float = 20.0,
custom_x_pct: float = 0.0,
custom_w_pct: float = 100.0
) -> Optional[Dict[str, Any]]:
"""
Runs Stage 0 -> Stage A -> Stage C.
Returns parsed subtitle blocks (original & translated) for Web Subtitle Editor.
"""
v_path = Path(video_path)
if not v_path.exists():
self._log(f"❌ Video not found: {video_path}")
return None
video_stem = v_path.stem
vtd = self.temp_dir / video_stem
vtd.mkdir(parents=True, exist_ok=True)
self.progress_fn(5, "STAGE_0_PREPARE")
calc_region = self.calculate_region(
v_path, region_preset, custom_y_pct, custom_h_pct, custom_x_pct, custom_w_pct
)
x, y, w, h = calc_region
# 1. Extract audio
extracted_audio = vtd / "extracted_audio.wav"
self._log("⚡ Trích xuất âm thanh từ video gốc...")
cmd = [
str(self.ffmpeg_path), "-y", "-i", str(v_path),
"-vn", "-acodec", "pcm_s16le", "-ar", "16000", "-ac", "1",
str(extracted_audio)
]
subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True)
# 2. OCR / ASR
self.progress_fn(25, "STAGE_A_OCR_ASR")
original_srt = vtd / "original.srt"
if mode == "ocr":
lang_label = "Tiếng Trung 🇨🇳" if str(source_lang).lower().startswith("zh") else "Tiếng Anh 🇺🇸"
self._log(f"👁️ Quét chữ phụ đề bằng AI Vision Models (ưu tiên {lang_label} - source_lang={source_lang})...")
ocr_box = (x, y, w, h) if (w > 0 and h > 0) else None
ok = self.ocr_engine.scan_video_subtitles_to_srt(str(v_path), str(original_srt), blur_region=ocr_box, source_lang=source_lang)
if not ok or not original_srt.exists() or original_srt.stat().st_size < 10:
# Report detailed reason before fallback
reason = []
if not ok:
reason.append("Vision API trả về rỗng / lỗi key")
if not original_srt.exists():
reason.append("SRT file không tồn tại")
elif original_srt.stat().st_size < 10:
reason.append(f"SRT quá nhỏ ({original_srt.stat().st_size} bytes)")
self._log(f"⚠️ OCR không phát hiện chữ ({', '.join(reason)}) -> Tự động chuyển sang Cloud ASR (fallback)...")
self._log(f" 💡 Kiểm tra: 1) API Key Gemini/OpenRouter còn hạn? 2) Vùng cắt [{x},{y},{w},{h}] có đúng chứa sub? 3) Thử tăng độ nhạy hoặc chọn preset Đáy Màn Hình")
ok = self.asr_engine.transcribe_audio_to_srt(str(extracted_audio), str(original_srt), source_lang=source_lang)
else:
self._log("🎙️ Nhận diện giọng nói bằng Cloud ASR (Groq Whisper Large-v3)...")
ok = self.asr_engine.transcribe_audio_to_srt(str(extracted_audio), str(original_srt), source_lang=source_lang)
if not ok or not original_srt.exists() or original_srt.stat().st_size < 10:
raise RuntimeError("Không thể trích xuất phụ đề từ video.")
# 3. Translation
self.progress_fn(50, "STAGE_C_TRANSLATION")
self._log("🌐 Dịch thuật bằng AI Translation Matrix...")
translated_srt = vtd / "translated.srt"
self._translate_srt_cloud(original_srt, translated_srt, source_lang=source_lang)
orig_blocks = self._parse_srt_blocks(original_srt.read_text(encoding="utf-8", errors="ignore"))
trans_blocks = self._parse_srt_blocks(translated_srt.read_text(encoding="utf-8", errors="ignore"))
combined = []
for ob in orig_blocks:
tb = next((t for t in trans_blocks if t["id"] == ob["id"]), None)
combined.append({
"id": ob["id"],
"timing": ob["timing"],
"start_ms": ob["start_ms"],
"end_ms": ob["end_ms"],
"original_text": ob["text"],
"vietnamese_text": tb["text"] if tb else ob["text"],
"sfx": ""
})
return {
"video_stem": video_stem,
"blocks": combined,
"original_srt": str(original_srt),
"translated_srt": str(translated_srt)
}
def render_from_subtitles(
self,
video_path: str,
subtitles_data: List[Dict[str, Any]],
voice: str = "vi-VN-NamMinhNeural",
speed: float = 1.0,
pitch: int = 0,
volume: int = 100,
region_preset: str = "custom",
sub_mask_mode: str = "box",
sub_style: str = "motion_drip",
custom_y_pct: float = 75.0,
custom_h_pct: float = 20.0,
custom_x_pct: float = 0.0,
custom_w_pct: float = 100.0,
ducking_ratio: float = 0.18,
enable_sfx: bool = True,
mute_original_audio: bool = False,
# ── Logo overlay ──
logo_enabled: bool = False,
logo_path: str = "",
logo_preset: str = "bottom_right",
logo_scale: float = 15.0,
logo_opacity: float = 1.0,
logo_x_pct: float = 80.0,
logo_y_pct: float = 80.0,
logo_chromakey: bool = True,
# ── CTA video overlay (every 5s) ──
cta_enabled: bool = False,
cta_path: str = "",
cta_preset: str = "bottom_center",
cta_scale: float = 35.0,
cta_opacity: float = 1.0,
cta_x_pct: float = 50.0,
cta_y_pct: float = 85.0,
cta_interval: float = 5.0,
cta_duration: float = 2.0,
cta_chromakey: bool = True
) -> Optional[str]:
"""Completes TTS synthesis, Ducking audio mix, and video rendering from user-reviewed subtitles."""
v_path = Path(video_path)
if not v_path.exists():
return None
video_stem = v_path.stem
vtd = self.temp_dir / video_stem
vtd.mkdir(parents=True, exist_ok=True)
start_time = time.time()
calc_region = self.calculate_region(
v_path, region_preset, custom_y_pct, custom_h_pct, custom_x_pct, custom_w_pct
)
x, y, w, h = calc_region
# Write edited SRT
translated_srt = vtd / "translated.srt"
srt_lines = []
for item in subtitles_data:
srt_lines.append(str(item["id"]))
srt_lines.append(item["timing"])
vi_text = self.normalizer.normalize(item["vietnamese_text"]) if hasattr(self.normalizer, "normalize") else item["vietnamese_text"]
srt_lines.append(vi_text)
srt_lines.append("")
translated_srt.write_text("\n".join(srt_lines), encoding="utf-8")
# 4. Cloud TTS
self.progress_fn(65, "STAGE_TTS_DUBBING")
self._log(f"🎙️ Tạo giọng đọc lồng tiếng ({voice}, speed={speed}x, volume={volume}%)...")
dubbing_wav = vtd / "dubbing.wav"
ok_tts = self.tts_engine.synthesize_srt_to_audio(
str(translated_srt),
str(dubbing_wav),
voice=voice,
speed=speed,
pitch=pitch,
volume=volume,
temp_dir=str(vtd / "tts_segments")
)
if not ok_tts or not dubbing_wav.exists():
raise RuntimeError("Lỗi tạo giọng đọc TTS.")
# 5. SFX & Audio Ducking
self.progress_fn(80, "STAGE_D_AUDIO_MIX")
extracted_audio = vtd / "extracted_audio.wav"
if not extracted_audio.exists():
cmd = [
str(self.ffmpeg_path), "-y", "-i", str(v_path),
"-vn", "-acodec", "pcm_s16le", "-ar", "16000", "-ac", "1",
str(extracted_audio)
]
subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True)
mixed_audio = vtd / "mixed_final.wav"
video_dur_sec = self._get_video_duration_sec(v_path)
if video_dur_sec > 0:
self._log(f"⏱️ Video gốc dài {video_dur_sec:.1f}s — audio mix sẽ được pad đúng bằng, chống cắt ngọn bởi -shortest.")
if mute_original_audio:
self._log("🔇 Tắt hoàn toàn tiếng gốc — chỉ giữ giọng Việt (mute_original_audio=ON)")
# Dubbing wav đã được time-stretch canvas đúng duration, chỉ cần chuẩn hóa sample-rate
cmd_mix = [
str(self.ffmpeg_path), "-y",
"-i", str(dubbing_wav),
"-ac", "2", "-ar", "48000",
str(mixed_audio)
]
subprocess.run(cmd_mix, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True)
# FIX 2p->20s: pad im lặng tới đúng video duration để -shortest không cắt video
if video_dur_sec > 0:
self._pad_audio_to_video_duration(mixed_audio, video_dur_sec)
else:
self._log("🎚️ Trộn nhạc nền (Auto Ducking) + Giọng đọc thuyết minh...")
filter_complex = (
f"[0:a]volume={ducking_ratio}[bg];"
f"[1:a]volume=1.0[dub];"
f"[bg][dub]amix=inputs=2:duration=longest:dropout_transition=2:normalize=0[aout]"
)
cmd_mix = [
str(self.ffmpeg_path), "-y",
"-i", str(extracted_audio),
"-i", str(dubbing_wav),
"-filter_complex", filter_complex,
"-map", "[aout]",
"-ac", "2", "-ar", "48000",
str(mixed_audio)
]
subprocess.run(cmd_mix, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True)
# Safety: amix duration=longest lẽ ra đã full, nhưng pad lại cho chắc
if video_dur_sec > 0:
self._pad_audio_to_video_duration(mixed_audio, video_dur_sec)
# 6. Render Video
self.progress_fn(90, "STAGE_D_RENDER")
self._log("🎬 Render video hoàn thiện: Xóa sub cũ & Đè sub tiếng Việt...")
final_output = self.output_dir / f"studio_final_{v_path.name}"
# Probe dimensions with cv2 then ffprobe fallback (handles vertical & cv2-missing envs)
vw, vh = 1920, 1080
try:
cap = cv2.VideoCapture(str(v_path))
w_tmp = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH)) or 0
h_tmp = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT)) or 0
cap.release()
if w_tmp > 0 and h_tmp > 0:
vw, vh = w_tmp, h_tmp
else:
raise ValueError("cv2 returned 0")
except Exception:
try:
import shutil as _sh
_ffprobe = _sh.which("ffprobe") or str(Path(self.ffmpeg_path).parent / "ffprobe.exe") if Path(self.ffmpeg_path).exists() else "ffprobe"
if not Path(_ffprobe).exists():
_ffprobe = "ffprobe"
_res = subprocess.run([_ffprobe, "-v", "error", "-select_streams", "v:0", "-show_entries", "stream=width,height", "-of", "json", str(v_path)], capture_output=True, text=True, timeout=5)
_j = json.loads(_res.stdout or "{}")
_ws = _j.get("streams", [{}])[0]
if _ws.get("width") and _ws.get("height"):
vw, vh = int(_ws["width"]), int(_ws["height"])
except Exception:
pass
translated_ass = vtd / "translated.ass"
self._convert_srt_to_ass(translated_srt, translated_ass, vw, vh, x, y, w, h, sub_style)
ass_escaped = str(translated_ass).replace("\\", "/").replace(":", "\\:")
sub_filter = f"subtitles='{ass_escaped}'"
vf_filters = []
if w > 0 and h > 0:
if sub_mask_mode == "delogo":
safe_x = max(2, x)
safe_y = max(2, y)
safe_w = max(4, min(w, vw - safe_x - 2))
safe_h = max(4, min(h, vh - safe_y - 2))
vf_filters.append(f"delogo=x={safe_x}:y={safe_y}:w={safe_w}:h={safe_h}")
elif sub_mask_mode == "box":
vf_filters.append(f"drawbox=x={x}:y={y}:w={w}:h={h}:color=black@0.92:t=fill")
# Determine logo & CTA overlay enabled (with fallback to bundled assets)
_logo_enabled = bool(logo_enabled and logo_path and Path(logo_path).exists())
if logo_enabled and not _logo_enabled:
_fb = self.base_dir / "assets" / "logo_ins_drippy4.png"
if _fb.exists():
_logo_enabled = True
logo_path = str(_fb)
_logo_path = Path(logo_path) if _logo_enabled else None
_cta_enabled = bool(cta_enabled and cta_path and Path(cta_path).exists())
if cta_enabled and not _cta_enabled:
_fb2 = self.base_dir / "assets" / "cta_ins_drippy4.mp4"
if _fb2.exists():
_cta_enabled = True
cta_path = str(_fb2)
_cta_path = Path(cta_path) if _cta_enabled else None
# If any overlay enabled, we need filter_complex
if _logo_enabled or _cta_enabled:
if _logo_enabled:
self._log(f"🖼️ Overlay logo: {Path(logo_path).name} preset={logo_preset} scale={logo_scale}% opacity={logo_opacity} chromakey={logo_chromakey}")
if _cta_enabled:
self._log(f"🎬 Overlay CTA: {Path(cta_path).name} preset={cta_preset} scale={cta_scale}% every {cta_interval}s for {cta_duration}s chromakey={cta_chromakey}")
# Build filter parts step-by-step. Inputs: 0:video, 1:audio, 2:logo(if), 3:cta(if)
# Determine indices
_logo_idx = 2 if _logo_enabled else None
_cta_idx = None
if _cta_enabled:
_cta_idx = 3 if _logo_enabled else 2
# Video preprocessing (delogo/box) -> [base]
_filter_parts = []
if vf_filters and len([f for f in vf_filters if f != sub_filter]) > 0:
_pre_vf = ",".join([f for f in vf_filters if f != sub_filter])
_filter_parts.append(f"[0:v]{_pre_vf}[base]")
_cur = "base"
else:
_filter_parts.append("[0:v]null[base]")
_cur = "base"
# Logo overlay
if _logo_enabled:
_logo_w_orig, _logo_h_orig = 2400, 1792
try:
_t = cv2.imread(str(_logo_path), cv2.IMREAD_UNCHANGED)
if _t is not None:
_logo_h_orig, _logo_w_orig = _t.shape[:2]
except Exception:
pass
_target_w = vw * float(logo_scale) / 100.0
_sf = max(0.02, min(0.5, _target_w / float(max(1, _logo_w_orig))))
_logo_vf_parts = []
if logo_chromakey:
_logo_vf_parts.append("colorkey=0x00FF00:0.3:0.1")
_logo_vf_parts.append("format=rgba")
_logo_vf_parts.append(f"scale=iw*{_sf:.4f}:ih*{_sf:.4f}:flags=lanczos")
if float(logo_opacity) < 0.99:
_logo_vf_parts.append(f"colorchannelmixer=aa={float(logo_opacity):.2f}")
_logo_vf = ",".join(_logo_vf_parts)
_m = 2.0
if logo_preset == "top_left":
_lx, _ly = f"W*{_m/100:.3f}", f"H*{_m/100:.3f}"
elif logo_preset == "top_right":
_lx, _ly = f"W-w-W*{_m/100:.3f}", f"H*{_m/100:.3f}"
elif logo_preset == "bottom_left":
_lx, _ly = f"W*{_m/100:.3f}", f"H-h-H*{_m/100:.3f}"
elif logo_preset == "bottom_right":
_lx, _ly = f"W-w-W*{_m/100:.3f}", f"H-h-H*{_m/100:.3f}"
elif logo_preset == "center":
_lx, _ly = "(W-w)/2", "(H-h)/2"
elif logo_preset == "top_center":
_lx, _ly = "(W-w)/2", f"H*{_m/100:.3f}"
elif logo_preset == "bottom_center":
_lx, _ly = "(W-w)/2", f"H-h-H*{_m/100:.3f}"
elif logo_preset == "custom":
_lx, _ly = f"W*{float(logo_x_pct)/100:.4f}", f"H*{float(logo_y_pct)/100:.4f}"
else:
_lx, _ly = f"W-w-W*{_m/100:.3f}", f"H-h-H*{_m/100:.3f}"
_filter_parts.append(f"[{_logo_idx}:v]{_logo_vf}[logo]")
_next = "with_logo" if _cta_enabled or sub_filter else "v"
_filter_parts.append(f"[{_cur}][logo]overlay={_lx}:{_ly}:format=rgb[{_next}]")
_cur = _next
# CTA overlay (periodic every interval)
if _cta_enabled:
# Probe CTA size
_cta_w_orig, _cta_h_orig = 1080, 1920
try:
_cap = cv2.VideoCapture(str(_cta_path))
_cta_w_orig = int(_cap.get(cv2.CAP_PROP_FRAME_WIDTH)) or _cta_w_orig
_cta_h_orig = int(_cap.get(cv2.CAP_PROP_FRAME_HEIGHT)) or _cta_h_orig
_cap.release()
except Exception:
pass
# TikTok CTA bump: enforce minimum 42% width for mobile visibility
effective_cta_scale = max(float(cta_scale), 42.0)
_cta_target_w = vw * effective_cta_scale / 100.0
_cta_sf = max(0.05, min(1.0, _cta_target_w / float(max(1, _cta_w_orig))))
# Clamp height to 80% of video height to avoid overflow (CTA vertical on horizontal video)
_cta_target_h = _cta_h_orig * _cta_sf
_max_h = vh * 0.90
if _cta_target_h > _max_h:
_cta_sf = _max_h / float(max(1, _cta_h_orig))
_cta_target_w = _cta_w_orig * _cta_sf
_cta_vf_parts = []
if cta_chromakey:
_cta_vf_parts.append("colorkey=0x00FF00:0.3:0.1")
_cta_vf_parts.append("format=rgba")
_cta_vf_parts.append(f"scale=iw*{_cta_sf:.4f}:ih*{_cta_sf:.4f}:flags=lanczos")
if float(cta_opacity) < 0.99:
_cta_vf_parts.append(f"colorchannelmixer=aa={float(cta_opacity):.2f}")
_cta_vf = ",".join(_cta_vf_parts)
_m2 = 2.0
if cta_preset == "top_left":
_cx, _cy = f"W*{_m2/100:.3f}", f"H*{_m2/100:.3f}"
elif cta_preset == "top_right":
_cx, _cy = f"W-w-W*{_m2/100:.3f}", f"H*{_m2/100:.3f}"
elif cta_preset == "bottom_left":
_cx, _cy = f"W*{_m2/100:.3f}", f"H-h-H*{_m2/100:.3f}"
elif cta_preset == "bottom_right":
_cx, _cy = f"W-w-W*{_m2/100:.3f}", f"H-h-H*{_m2/100:.3f}"
elif cta_preset == "center":
_cx, _cy = "(W-w)/2", "(H-h)/2"
elif cta_preset == "top_center":
_cx, _cy = "(W-w)/2", f"H*{_m2/100:.3f}"
elif cta_preset == "bottom_center":
_cx, _cy = "(W-w)/2", f"H-h-H*{_m2/100:.3f}"
elif cta_preset == "custom":
_cx, _cy = f"W*{float(cta_x_pct)/100:.4f}", f"H*{float(cta_y_pct)/100:.4f}"
else:
_cx, _cy = "(W-w)/2", f"H-h-H*{_m2/100:.3f}"
_enable = f"lt(mod(t\\,{float(cta_interval)})\\,{float(cta_duration)})"
_filter_parts.append(f"[{_cta_idx}:v]{_cta_vf}[cta]")
_next2 = "v" if not sub_filter else "with_cta"
# Use escaped comma for FFmpeg enable expression
_filter_parts.append(f"[{_cur}][cta]overlay={_cx}:{_cy}:format=rgb:enable='{_enable}'[{_next2}]")
_cur = _next2
# TikTok polish: CFR 30 + even dims + light sharpen BEFORE subtitles (keeps text razor sharp)
tiktok_polish = "fps=30:round=near,scale=trunc(iw/2)*2:trunc(ih/2)*2:flags=lanczos+accurate_rnd+full_chroma_int:sws_dither=ed,unsharp=3:3:0.35:3:3:0.0"
if sub_filter:
_filter_parts.append(f"[{_cur}]{tiktok_polish}[polished]")
_cur = "polished"
_filter_parts.append(f"[{_cur}]{sub_filter}[v]")
_cur = "v"
else:
_filter_parts.append(f"[{_cur}]{tiktok_polish}[v]")
_cur = "v"
_filter_complex = ";".join(_filter_parts)
# Build ffmpeg inputs: video, audio, logo(if), cta(if) - add -shortest when looped overlays present
cmd_inputs = [str(self.ffmpeg_path), "-y", "-fflags", "+genpts", "-avoid_negative_ts", "make_zero", "-i", str(v_path), "-i", str(mixed_audio)]
if _logo_enabled:
cmd_inputs += ["-loop", "1", "-i", str(_logo_path)]
if _cta_enabled:
cmd_inputs += ["-stream_loop", "999", "-i", str(_cta_path)]
_extra = ["-shortest"] if (_logo_enabled or _cta_enabled) else []
# TikTok spec: High Profile yuv420p, 30 CFR, 48k AAC, faststart, accurate sync
tiktok_video_args = ["-r", "30", "-c:v", "libx264", "-preset", "medium", "-crf", "18", "-profile:v", "high", "-level", "4.1", "-pix_fmt", "yuv420p", "-g", "60", "-keyint_min", "30", "-sc_threshold", "0", "-x264-params", "ref=4:bframes=2:me=hex:subme=7:psy=1:psy-rd=0.8:aq-mode=2", "-colorspace", "bt709", "-color_primaries", "bt709", "-color_trc", "bt709", "-color_range", "tv"]
tiktok_audio_args = ["-c:a", "aac", "-profile:a", "aac_low", "-ar", "48000", "-ac", "2", "-b:a", "192k", "-af", "aresample=async=1:min_hard_comp=0.100000:first_pts=0"]
tiktok_mux_args = ["-movflags", "+faststart", "-fflags", "+genpts", "-max_interleave_delta", "100M", "-vsync", "cfr", "-fps_mode", "cfr"]
cmd_render = cmd_inputs + ["-filter_complex", _filter_complex, "-map", "[v]", "-map", "1:a:0"] + _extra + tiktok_video_args + tiktok_audio_args + tiktok_mux_args + ["-shortest", str(final_output)]
res = subprocess.run(cmd_render, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True)
if res.returncode != 0:
self._log(f"⚠️ Overlay render failed ({res.stderr[:300]}), fallback to TikTok-spec normal render...")
# Rebuild VF with TikTok polish inserted before subs (same as non-overlay branch)
tiktok_polish_fb = "fps=30:round=near,scale=trunc(iw/2)*2:trunc(ih/2)*2:flags=lanczos+accurate_rnd+full_chroma_int:sws_dither=ed,unsharp=3:3:0.35:3:3:0.0"
vf_without_sub_fb = [f for f in vf_filters if f != sub_filter]
ordered_fb = []
if vf_without_sub_fb:
ordered_fb.extend(vf_without_sub_fb)
ordered_fb.append(tiktok_polish_fb)
ordered_fb.append(sub_filter)
final_vf = ",".join(ordered_fb)
cmd_render = [
str(self.ffmpeg_path), "-y",
"-fflags", "+genpts", "-avoid_negative_ts", "make_zero",
"-i", str(v_path),
"-i", str(mixed_audio),
"-vf", final_vf,
"-map", "0:v:0", "-map", "1:a:0",
"-r", "30",
"-c:v", "libx264", "-preset", "medium", "-crf", "18",
"-profile:v", "high", "-level", "4.1", "-pix_fmt", "yuv420p",
"-g", "60", "-keyint_min", "30", "-sc_threshold", "0",
"-x264-params", "ref=4:bframes=2:me=hex:subme=7:psy=1:psy-rd=0.8:aq-mode=2",
"-colorspace", "bt709", "-color_primaries", "bt709", "-color_trc", "bt709", "-color_range", "tv",
"-c:a", "aac", "-profile:a", "aac_low", "-ar", "48000", "-ac", "2", "-b:a", "192k",
"-af", "aresample=async=1:min_hard_comp=0.100000:first_pts=0",
"-movflags", "+faststart",
"-fflags", "+genpts", "-max_interleave_delta", "100M",
"-vsync", "cfr", "-fps_mode", "cfr",
"-shortest",
str(final_output)
]
res = subprocess.run(cmd_render, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True)
if res.returncode != 0:
self._log(f"⚠️ Fallback render không mask: {res.stderr[:120]}")
fallback_cmd = [
str(self.ffmpeg_path), "-y",
"-fflags", "+genpts",
"-i", str(v_path),
"-i", str(mixed_audio),
"-vf", f"{tiktok_polish_fb},{sub_filter}",
"-map", "0:v:0", "-map", "1:a:0",
"-r", "30",
"-c:v", "libx264", "-preset", "medium", "-crf", "18",
"-profile:v", "high", "-pix_fmt", "yuv420p",
"-c:a", "aac", "-profile:a", "aac_low", "-ar", "48000", "-ac", "2", "-b:a", "192k",
"-movflags", "+faststart",
"-shortest",
str(final_output)
]
subprocess.run(fallback_cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True)
else:
# TikTok polish inserted before subtitles for sharpness + CFR + even dims
tiktok_polish = "fps=30:round=near,scale=trunc(iw/2)*2:trunc(ih/2)*2:flags=lanczos+accurate_rnd+full_chroma_int:sws_dither=ed,unsharp=3:3:0.35:3:3:0.0"
# Order: mask (delogo/box) -> polish -> subtitles
vf_without_sub = [f for f in vf_filters if f != sub_filter]
# Rebuild vf in TikTok-optimal order
ordered_vf = []
if vf_without_sub:
ordered_vf.extend(vf_without_sub)
ordered_vf.append(tiktok_polish)
ordered_vf.append(sub_filter)
final_vf = ",".join(ordered_vf)
cmd_render = [
str(self.ffmpeg_path), "-y",
"-fflags", "+genpts", "-avoid_negative_ts", "make_zero",
"-i", str(v_path),
"-i", str(mixed_audio),
"-vf", final_vf,
"-map", "0:v:0", "-map", "1:a:0",
"-r", "30",
"-c:v", "libx264", "-preset", "medium", "-crf", "18",
"-profile:v", "high", "-level", "4.1", "-pix_fmt", "yuv420p",
"-g", "60", "-keyint_min", "30", "-sc_threshold", "0",
"-x264-params", "ref=4:bframes=2:me=hex:subme=7:psy=1:psy-rd=0.8:aq-mode=2",
"-colorspace", "bt709", "-color_primaries", "bt709", "-color_trc", "bt709", "-color_range", "tv",
"-c:a", "aac", "-profile:a", "aac_low", "-ar", "48000", "-ac", "2", "-b:a", "192k",
"-af", "aresample=async=1:min_hard_comp=0.100000:first_pts=0",
"-movflags", "+faststart",
"-fflags", "+genpts", "-max_interleave_delta", "100M",
"-vsync", "cfr", "-fps_mode", "cfr",
"-shortest",
str(final_output)
]
res = subprocess.run(cmd_render, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True)
if res.returncode != 0:
self._log(f"⚠️ Fallback render không mask: {res.stderr[:120]} | retry with minimal TikTok spec")
# Minimal fallback: at least ensure TikTok audio/video spec
fallback_cmd = [
str(self.ffmpeg_path), "-y",
"-fflags", "+genpts",
"-i", str(v_path),
"-i", str(mixed_audio),
"-vf", f"{tiktok_polish},{sub_filter}",
"-map", "0:v:0", "-map", "1:a:0",
"-r", "30",
"-c:v", "libx264", "-preset", "medium", "-crf", "18",
"-profile:v", "high", "-pix_fmt", "yuv420p",
"-c:a", "aac", "-profile:a", "aac_low", "-ar", "48000", "-ac", "2", "-b:a", "192k",
"-movflags", "+faststart",
"-shortest",
str(final_output)
]
subprocess.run(fallback_cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True)
total_elapsed = time.time() - start_time
# Verify output duration vs input — báo động ngay nếu bị cắt ngọn (2p->20s)
try:
out_dur = self._get_video_duration_sec(final_output)
in_dur = video_dur_sec or self._get_video_duration_sec(v_path)
if in_dur > 0 and out_dur > 0:
if out_dur < in_dur - 3.0:
self._log(f"❌ CẢNH BÁO ĐỘ DÀI: input {in_dur:.1f}s nhưng output chỉ {out_dur:.1f}s "
f"(thiếu {in_dur - out_dur:.1f}s)! Kiểm tra phụ đề/ASR-OCR có bị rụng đuôi không.")
else:
self._log(f"⏱️ Verify độ dài OK: input {in_dur:.1f}s -> output {out_dur:.1f}s.")
except Exception:
pass
self.progress_fn(100, "DONE")
self._log(f"🎉 HOÀN THÀNH XUẤT SẮC TRONG {total_elapsed:.1f}s!")
self._log(f"📁 Video đã lưu tại: {final_output}")
return str(final_output)
def run_video(
self,
video_path: str,
mode: str = "asr",
source_lang: str = "zh",
voice: str = "vi-VN-NamMinhNeural",
speed: float = 1.0,
pitch: int = 0,
volume: int = 100,
region_preset: str = "custom",
sub_mask_mode: str = "box",
sub_style: str = "motion_drip",
custom_y_pct: float = 75.0,
custom_h_pct: float = 20.0,
custom_x_pct: float = 0.0,
custom_w_pct: float = 100.0,
ducking_ratio: float = 0.18,
enable_sfx: bool = True,
mute_original_audio: bool = False,
logo_enabled: bool = False,
logo_path: str = "",
logo_preset: str = "bottom_right",
logo_scale: float = 15.0,
logo_opacity: float = 1.0,
logo_x_pct: float = 80.0,
logo_y_pct: float = 80.0,
logo_chromakey: bool = True,
cta_enabled: bool = False,
cta_path: str = "",
cta_preset: str = "bottom_center",
cta_scale: float = 35.0,
cta_opacity: float = 1.0,
cta_x_pct: float = 50.0,
cta_y_pct: float = 85.0,
cta_interval: float = 5.0,
cta_duration: float = 2.0,
cta_chromakey: bool = True
) -> Optional[str]:
"""Full 1-Click End-to-End Automatic Dubbing Pipeline."""
v_path = Path(video_path)
if not v_path.exists():
self._log(f"❌ Video not found: {video_path}")
return None
# Register into SQLite Job DB
try:
self.job_manager.register_job(str(v_path))
except Exception:
pass
try:
# 1. Extract subtitles
subs_res = self.extract_subtitles_only(
video_path=video_path,
mode=mode,
source_lang=source_lang,
region_preset=region_preset,
custom_y_pct=custom_y_pct,
custom_h_pct=custom_h_pct,
custom_x_pct=custom_x_pct,
custom_w_pct=custom_w_pct
)
if not subs_res:
return None
# 2. Render from subtitles
return self.render_from_subtitles(
video_path=video_path,
subtitles_data=subs_res["blocks"],
voice=voice,
speed=speed,
pitch=pitch,
volume=volume,
region_preset=region_preset,
sub_mask_mode=sub_mask_mode,
sub_style=sub_style,
custom_y_pct=custom_y_pct,
custom_h_pct=custom_h_pct,
custom_x_pct=custom_x_pct,
custom_w_pct=custom_w_pct,
ducking_ratio=ducking_ratio,
enable_sfx=enable_sfx,
mute_original_audio=mute_original_audio,
logo_enabled=logo_enabled,
logo_path=logo_path,
logo_preset=logo_preset,
logo_scale=logo_scale,
logo_opacity=logo_opacity,
logo_x_pct=logo_x_pct,
logo_y_pct=logo_y_pct,
logo_chromakey=logo_chromakey,
cta_enabled=cta_enabled,
cta_path=cta_path,
cta_preset=cta_preset,
cta_scale=cta_scale,
cta_opacity=cta_opacity,
cta_x_pct=cta_x_pct,
cta_y_pct=cta_y_pct,
cta_interval=cta_interval,
cta_duration=cta_duration,
cta_chromakey=cta_chromakey
)
except Exception as e:
self._log(f"❌ LỖI PIPELINE: {str(e)}")
self.progress_fn(0, "FAILED")
return None
def _convert_srt_to_ass(
self,
srt_path: Path,
ass_path: Path,
vw: int,
vh: int,
box_x: int,
box_y: int,
box_w: int,
box_h: int,
sub_style: str = "motion_drip"
):
"""Converts SRT to styled ASS format for TikTok/Shorts typography."""
content = srt_path.read_text(encoding="utf-8", errors="ignore")
# TikTok safe zone: keep subtitle inside 82% height, 7% bottom margin, larger font for mobile legibility
safe_margin_v = max(42, int(vh * 0.07))
raw_margin_v = int(vh - (box_y + box_h * 0.75)) if box_h > 0 else safe_margin_v
margin_v = max(safe_margin_v, raw_margin_v)
# Clamp to avoid bottom UI overlap
max_bottom = int(vh * 0.12)
if margin_v < max_bottom:
margin_v = max_bottom
# Larger font for TikTok: vertical 56-72, horizontal 48-64
if box_h > 0:
base = int(box_h * 0.52)
if vh > vw: # vertical TikTok
font_size = max(42, min(72, base))
else:
font_size = max(38, min(64, base))
else:
font_size = 58 if vh > vw else 50
# Typography Color & Outline palettes - TikTok optimized for high compression
if sub_style == "motion_drip" or sub_style == "yellow_bold":
font_color = "&H0000FFFF" # Bright Yellow
outline_color = "&H00000000" # Pure Black
back_color = "&H99000000"
outline = 5.0
shadow = 2.2
elif sub_style == "neon_cyan":
font_color = "&H00FFFF00" # Cyan
outline_color = "&H00000000"
back_color = "&H99000000"
outline = 5.0
shadow = 2.2
elif sub_style == "capsule_tag":
font_color = "&H00FFFFFF" # White
outline_color = "&H00111111"
back_color = "&HBB000000"
outline = 3.2
shadow = 0.0
else: # white_bold
font_color = "&H00FFFFFF" # Crisp White
outline_color = "&H00000000"
back_color = "&H99000000"
outline = 4.8
shadow = 2.0
ass_header = f"""[Script Info]
ScriptType: v4.00+
PlayResX: {vw}
PlayResY: {vh}
ScaledBorderAndShadow: yes
[V4+ Styles]
Format: Name, Fontname, Fontsize, PrimaryColour, SecondaryColour, OutlineColour, BackColour, Bold, Italic, Underline, StrikeOut, ScaleX, ScaleY, Spacing, Angle, BorderStyle, Outline, Shadow, Alignment, MarginL, MarginR, MarginV, Encoding
Style: Default,DejaVu Sans,{font_size},{font_color},&H000000FF,{outline_color},{back_color},1,0,0,0,100,100,0,0,1,{outline},{shadow},2,40,40,{margin_v},1
[Events]
Format: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text
"""
events = []
pattern = r"(\d+)\s+(\d{2}:\d{2}:\d{2}[.,]\d{3})\s*-->\s*(\d{2}:\d{2}:\d{2}[.,]\d{3})\s*\n(.*?)(?=\n\s*\d+\s+\d{2}:\d{2}:\d{2}[.,]\d{3}\s*-->|\Z)"
for m in re.finditer(pattern, content, re.DOTALL):
start_str = m.group(2).strip().replace(",", ".")
end_str = m.group(3).strip().replace(",", ".")
s_parts = start_str.split(":")
e_parts = end_str.split(":")
s_ass = f"{int(s_parts[0])}:{s_parts[1]}:{float(s_parts[2]):05.2f}"
e_ass = f"{int(e_parts[0])}:{e_parts[1]}:{float(e_parts[2]):05.2f}"
text = " ".join(line.strip() for line in m.group(4).splitlines() if line.strip())
if text:
# FIX: escape ký tự ASS đặc biệt — "{deal}", "\", emoticon bị libass
# hiểu thành override-tag -> render lỗi/hiện chữ thô
text = text.replace("\\", "\\\\").replace("{", "\\{").replace("}", "\\}")
events.append(f"Dialogue: 0,{s_ass},{e_ass},Default,,0,0,0,,{text}")
ass_path.write_text(ass_header + "\n".join(events), encoding="utf-8")
def _parse_srt_blocks(self, content: str) -> List[Dict]:
# FIX: giờ 1 chữ số (\d{1,2}) — timestamp "0:00:01,000" của Whisper/tool lẻ bị miss cả block
pattern = r"(\d+)\s+(\d{1,2}:\d{2}:\d{2}[.,]\d{3})\s*-->\s*(\d{1,2}:\d{2}:\d{2}[.,]\d{3})\s*\n(.*?)(?=\n\s*\d+\s+\d{1,2}:\d{2}:\d{2}[.,]\d{3}\s*-->|\Z)"
blocks = []
for m in re.finditer(pattern, content, re.DOTALL):
text = " ".join(line.strip() for line in m.group(4).splitlines() if line.strip())
start_str = m.group(2).strip()
end_str = m.group(3).strip()
if text:
blocks.append({
"id": int(m.group(1)),
"timing": f"{start_str} --> {end_str}",
"text": text,
"start_ms": self._ts_to_ms(start_str),
"end_ms": self._ts_to_ms(end_str),
})
return blocks
@staticmethod
def _ts_to_ms(ts: str) -> int:
ts = ts.strip().replace(".", ",")
m = re.match(r"(\d+):(\d+):(\d+)[,](\d+)", ts)
if m:
h, mins, s, ms = map(int, m.groups())
return ((h * 3600 + mins * 60 + s) * 1000) + ms
return 0
def _build_system_prompt(self) -> str:
glossary_sample = ", ".join([f"{k} -> {v}" for k, v in list(self.glossary.items())[:35]])
return (
"Bạn là chuyên gia dịch thuật video Sneaker, Thời trang Streetwear, Review sản phẩm từ tiếng Trung/Anh sang tiếng Việt tự nhiên, sành điệu, bắt trend Gen Z.\n"
"QUY TẮC BẮT BUỘC:\n"
"1. Dịch từng dòng theo cấu trúc: [N] Câu dịch tiếng Việt hoàn chỉnh.\n"
"2. Giữ nguyên thuật ngữ & thương hiệu tiếng Anh (Nike, Jordan, Yeezy, BAPE, Supreme, Rick Owens, outfit, fit, drip, full box, collab, signature...). \n"
f"3. Bắt buộc áp dụng từ điển chuyên ngành: {glossary_sample}\n"
"4. Dịch ĐẦY ĐỦ Ý NGHĨA trọn vẹn của câu, giữ đủ các từ khoá, TUYỆT ĐỐI KHÔNG cắt xén, không bỏ lửng hay cắt cụt mất từ ở đầu, giữa hoặc cuối câu. Câu dịch phải trọn vẹn ngữ nghĩa, tự nhiên và dễ hiểu.\n"
"5. KHÔNG giải thích, CHỈ trả về danh sách các dòng [N] Tiếng Việt."
)
def _translate_srt_cloud(self, srt_in: Path, srt_out: Path, source_lang: str = "zh"):
content = srt_in.read_text(encoding="utf-8", errors="ignore")
blocks = self._parse_srt_blocks(content)
if not blocks:
srt_out.write_text(content, encoding="utf-8")
return
system_prompt = self._build_system_prompt()
validator = TranslationValidator()
trans_map = {}
# Phase 1: Batch translation in chunks of 20
chunk_size = 20
for i in range(0, len(blocks), chunk_size):
chunk = blocks[i : i + chunk_size]
res = self._translate_chunk_with_retry(chunk, system_prompt, validator, max_retries=2)
trans_map.update(res)
# Phase 2: Retry missing / Chinese-leaked blocks
missing_blocks = [b for b in blocks if (b["id"] not in trans_map or has_chinese(trans_map.get(b["id"], "")))]
if missing_blocks:
self._log(f"🔄 Đang hoàn thiện nốt {len(missing_blocks)} câu dịch còn lại...")
retry_result = self._translate_chunk_with_retry(missing_blocks, system_prompt, validator, max_retries=2)
trans_map.update(retry_result)
# Phase 3: Google Translate fallback
still_missing = [b for b in blocks if (b["id"] not in trans_map or has_chinese(trans_map.get(b["id"], "")))]
if still_missing:
self._log(f"🌐 {len(still_missing)} câu cần fallback Google Translate...")
google_fb = GoogleTranslateFallback(log_fn=self._log)
google_result = google_fb.translate_blocks(still_missing)
for bid_str, text in google_result.items():
trans_map[int(bid_str)] = text
# Phase 4: Post-editing
self._log("✨ Chạy Post-Editor sửa lỗi dịch...")
for b in blocks:
bid = b["id"]
if bid in trans_map:
trans_map[bid] = post_edit_translation(b["text"], trans_map[bid])
# Phase 5: Quality Guard
self._log("🛡️ Quality Guard kiểm tra chất lượng bản dịch...")
guard = TranslationQualityGuard(min_score=70)
guard_dict = {str(b["id"]): trans_map.get(b["id"], b["text"]) for b in blocks}
fixed_dict, quality_report = guard.audit_and_fix(blocks, guard_dict)
for b in blocks:
bid_str = str(b["id"])
if bid_str in fixed_dict:
trans_map[b["id"]] = fixed_dict[bid_str]
# Phase 6: Subtitle text cleanup (safe formatting)
for b in blocks:
bid = b["id"]
if bid in trans_map:
t = str(trans_map[bid]).strip()
t = re.sub(r"\s+", " ", t).strip(" ,")
trans_map[bid] = t
# Phase 7: Output SRT
out_lines = []
for b in blocks:
vi_text = trans_map.get(b["id"], b["text"])
out_lines.append(str(b["id"]))
out_lines.append(b["timing"])
out_lines.append(vi_text)
out_lines.append("")
srt_out.write_text("\n".join(out_lines), encoding="utf-8")
self._log(f"✅ Dịch hoàn tất {len(blocks)} câu với 7 bước kiểm tra chất lượng.")
def _translate_chunk_with_retry(
self,
chunk: List[Dict],
system_prompt: str,
validator: TranslationValidator,
max_retries: int = 2
) -> Dict[int, str]:
prompt_lines = [f"[{b['id']}] {b['text']}" for b in chunk]
full_transcript = "\n".join(prompt_lines)
result = {}
for attempt in range(max_retries + 1):
raw_res = self._direct_25_model_translate(system_prompt, full_transcript)
cleaned_res = re.sub(r"<think>.*?</think>", "", raw_res, flags=re.DOTALL).strip()
attempt_map = {}
for line in cleaned_res.splitlines():
m = re.match(r"^\s*\[(\d+)\]\s*(.*)$", line.strip())
if m:
attempt_map[int(m.group(1))] = m.group(2).strip()
# FIX rụng đuôi dịch: chỉ accept early-return khi ĐỦ 100% số dòng chunk.
# Trước đây subset 5/20 dòng pass validate -> return thiếu 15 dòng im lặng.
expected_ids = {b["id"] for b in chunk}
got_ids = set(attempt_map.keys()) & expected_ids
val_dict = {str(b["id"]): attempt_map.get(b["id"], "") for b in chunk if b["id"] in attempt_map}
val_chunk = [b for b in chunk if b["id"] in attempt_map]
if val_chunk and val_dict:
is_valid, error_msg = validator.validate(val_chunk, val_dict)
if is_valid and got_ids == expected_ids:
result.update(attempt_map)
return result
if not is_valid:
self._log(f"⚠️ Chunk validate fail ({len(got_ids)}/{len(expected_ids)} dòng): {error_msg} — retry...")
elif got_ids != expected_ids:
missing = sorted(expected_ids - got_ids)
self._log(f"⚠️ Chunk thiếu {len(missing)}/{len(expected_ids)} dòng {missing[:8]} — retry...")
for bid, text in attempt_map.items():
src_block = next((b for b in chunk if b["id"] == bid), None)
if src_block:
ok, _ = validator.validate_single_block(bid, src_block["text"], text)
if ok:
result[bid] = text
elif attempt_map:
# Không parse được dòng nào theo format [N] — giữ lại để Phase 2/3 xử lý
self._log(f"⚠️ Chunk parse được 0/{len(expected_ids)} dòng hợp lệ — retry...")
return result
NIM_TRANSLATION_MODEL = "nvidia/nemotron-3.5-lightning-30b-a3b"
NIM_TRANSLATION_URL = "https://integrate.api.nvidia.com/v1/chat/completions"
def _get_nim_keys(self):
"""Lấy Nvidia NIM keys từ .env/env: NIM_API_KEY > NVIDIA_KEY_1..N."""
import os
keys = []
direct = (os.getenv("NIM_API_KEY", "") or "").strip()
if direct:
keys.append(direct)
# Ưu tiên engine đã load sẵn (.env + os.environ)
for src in (getattr(self.asr_engine, "nvidia_keys", []),
getattr(self.ocr_engine, "nvidia_keys", [])):
for k in (src or []):
if k and k not in keys:
keys.append(k)
# Quét NVIDIA_KEY_N trực tiếp từ environ (cho HF Secrets)
for i in range(1, 20):
v = (os.getenv(f"NVIDIA_KEY_{i}", "") or "").strip()
if v and v not in keys:
keys.append(v)
return keys
def _direct_25_model_translate(self, system_prompt: str, user_content: str) -> str:
messages = [
{"role": "system", "content": system_prompt},
{"role": "user", "content": user_content}
]
# 0. NVIDIA NIM Nemotron 3.5 Lightning 30B (ƯU TIÊN #1 — key có sẵn trong tool)
nim_keys = self._get_nim_keys()
for key in nim_keys:
try:
res = requests.post(
self.NIM_TRANSLATION_URL,
headers={"Authorization": f"Bearer {key}", "Content-Type": "application/json"},
json={"model": self.NIM_TRANSLATION_MODEL, "messages": messages,
"temperature": 0.2, "max_tokens": 4096},
timeout=30
)
if res.status_code == 200:
content = res.json()["choices"][0]["message"]["content"].strip()
content = re.sub(r"<think>.*?</think>", "", content, flags=re.DOTALL).strip()
if len(content) > 15:
self._log(f"✅ Dịch bằng NVIDIA NIM ({self.NIM_TRANSLATION_MODEL})")
return content
else:
self._log(f"⚠️ NIM {res.status_code}: {res.text[:120]}")
except Exception as e:
self._log(f"⚠️ NIM lỗi: {e}")
continue
# 1. Google Gemini (Fastest & highest accuracy for Asian languages)
gemini_keys = self.ocr_engine.gemini_keys
for model in ["gemini-3.6-flash", "gemini-3.5-flash", "gemini-flash-latest", "gemini-pro-latest", "gemini-2.5-flash", "gemini-2.0-flash"]:
for key in gemini_keys:
try:
url = f"https://generativelanguage.googleapis.com/v1beta/models/{model}:generateContent?key={key}"
payload = {
"contents": [{"parts": [{"text": f"{system_prompt}\n\n{user_content}"}]}],
"generationConfig": {"temperature": 0.2, "maxOutputTokens": 4096}
}
res = requests.post(url, json=payload, timeout=25)
if res.status_code == 200:
content = res.json()["candidates"][0]["content"]["parts"][0]["text"].strip()
if len(content) > 15:
return content
except Exception:
pass
# 2. Groq (High speed LLM)
groq_keys = self.asr_engine.groq_keys
for model in ["openai/gpt-oss-20b", "groq/compound", "qwen/qwen3.6-27b", "llama-3.3-70b-versatile", "llama-3.1-8b-instant"]:
for key in groq_keys:
try:
res = requests.post(
"https://api.groq.com/openai/v1/chat/completions",
headers={"Authorization": f"Bearer {key}", "Content-Type": "application/json"},
json={"model": model, "messages": messages, "temperature": 0.2, "max_tokens": 4096},
timeout=25
)
if res.status_code == 200:
content = res.json()["choices"][0]["message"]["content"].strip()
content = re.sub(r"<think>.*?</think>", "", content, flags=re.DOTALL).strip()
if len(content) > 15:
return content
except Exception:
pass
# 3. OpenRouter Free & Standard Models
or_keys = self.ocr_engine.openrouter_keys
for model in [
"nvidia/nemotron-3.5-lightning:free",
"inclusionai/ling-3.0-flash-fin:free",
"liquid/lfm-2.5-2.6b:free",
"deepseek/deepseek-r1:free",
"qwen/qwen-2.5-72b-instruct:free"
]:
for key in or_keys:
try:
res = requests.post(
"https://openrouter.ai/api/v1/chat/completions",
headers={"Authorization": f"Bearer {key}", "Content-Type": "application/json", "HTTP-Referer": "https://trungsangviet.local", "X-Title": "TrungSangViet"},
json={"model": model, "messages": messages, "temperature": 0.2, "max_tokens": 4096},
timeout=30
)
if res.status_code == 200:
content = res.json()["choices"][0]["message"]["content"].strip()
content = re.sub(r"<think>.*?</think>", "", content, flags=re.DOTALL).strip()
if len(content) > 15:
return content
except Exception:
pass
# FIX: trước đây fail hết model thì trả source (tiếng Trung) như "đã dịch" ->
# caller parse thành trans_map, có thể lọt qua validator vào SRT.
# Nay trả "" để Phase 2 retry + Phase 3 Google fallback làm việc đúng.
self._log("❌ Tất cả translation models đều fail cho chunk này — để trống cho fallback.")
return ""