import sys import os import re import json import argparse import subprocess from pathlib import Path # Enforce UTF-8 for Windows console if sys.platform == 'win32': try: if hasattr(sys.stdout, 'reconfigure'): sys.stdout.reconfigure(encoding='utf-8') if hasattr(sys.stderr, 'reconfigure'): sys.stderr.reconfigure(encoding='utf-8') except Exception: pass def check_nvenc_available(ffmpeg_path): cmd = [ str(ffmpeg_path), "-y", "-f", "lavfi", "-i", "color=c=black:s=256x256", "-t", "1", "-c:v", "h264_nvenc", "-f", "null", "-" ] try: startupinfo = None if sys.platform == 'win32': startupinfo = subprocess.STARTUPINFO() startupinfo.dwFlags |= subprocess.STARTF_USESHOWWINDOW res = subprocess.run( cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, encoding="utf-8", errors="ignore", startupinfo=startupinfo, timeout=10 ) return res.returncode == 0 except Exception: return False def _ass_time_from_srt(value): m = re.match(r"^(\d{2}):(\d{2}):(\d{2})[,.](\d{3})$", str(value).strip()) if not m: return "0:00:00.00" hh, mm, ss, ms = [int(x) for x in m.groups()] centis = int(round(ms / 10.0)) return f"{hh}:{mm:02d}:{ss:02d}.{centis:02d}" def _parse_srt_events(srt_path): try: content = Path(srt_path).read_text(encoding="utf-8", errors="ignore").strip().replace("\r\n", "\n") except Exception: return [] pattern = r"(\d+)\s+(\d{2}:\d{2}:\d{2}[.,]\d{3})\s*-->\s*(\d{2}:\d{2}:\d{2}[.,]\d{3})\s*\n(.*?)(?=\n\s*\d+\s+\d{2}:\d{2}:\d{2}[.,]\d{3}\s*-->|\Z)" events = [] for match in re.finditer(pattern, content, re.DOTALL): text = " ".join(line.strip() for line in match.group(4).splitlines() if line.strip()) if text: events.append((match.group(2), match.group(3), text)) return events def _ass_escape(text): return str(text or "").replace("\\", "\\\\").replace("{", r"\{").replace("}", r"\}").replace("\n", r"\N") def _shift_srt(srt_path, seconds): """Shift every timestamp in an SRT file by -seconds (in place). Used after trimming leading silence so hardsub captions stay aligned with the video.""" srt_path = Path(srt_path) if seconds <= 0 or not srt_path.exists(): return content = srt_path.read_text(encoding="utf-8", errors="ignore").replace("\r\n", "\n") def shift_ts(m): def to_ms(t): tm = re.match(r"(\d+):(\d+):(\d+)[.,](\d+)", t) if not tm: return 0 hh, mm, ss, ms = (int(x) for x in tm.groups()) return ((hh * 3600 + mm * 60 + ss) * 1000) + ms def from_ms(v): v = max(0, int(v)) hh, rem = divmod(v, 3600000) mm, rem = divmod(rem, 60000) ss, ms = divmod(rem, 1000) return f"{hh:02d}:{mm:02d}:{ss:02d},{ms:03d}" delta = int(round(seconds * 1000)) return f"{from_ms(to_ms(m.group(1)) - delta)} --> {from_ms(to_ms(m.group(2)) - delta)}" pattern = re.compile(r"(\d{2}:\d{2}:\d{2}[.,]\d{3})\s*-->\s*(\d{2}:\d{2}:\d{2}[.,]\d{3})") new_content = pattern.sub(shift_ts, content) srt_path.write_text(new_content, encoding="utf-8") def _wrap_caption_words(text, max_chars=22, max_lines=3): """Wrap caption text naturally across lines without dropping words.""" words = str(text or "").split() if not words: return [] lines = [] current = "" for word in words: candidate = word if not current else f"{current} {word}" if len(candidate) <= max_chars or not current: current = candidate else: if len(lines) < max_lines - 1: lines.append(current) current = word else: # Ở dòng cuối cùng, tiếp tục gộp toàn bộ từ còn lại vào dòng này, không được bỏ sót current = f"{current} {word}" if current: lines.append(current) # Xử lý các từ nối không đứng lơ lửng một mình ở cuối dòng connectors = {"và", "với", "của", "là", "thì", "mà", "rằng", "nên", "nhưng", "hoặc", "vì"} if len(lines) > 1: for idx in range(len(lines) - 1): parts = lines[idx].split() if len(parts) > 1 and parts[-1].strip(" ,.!?;:").lower() in connectors: moved_word = parts.pop() lines[idx] = " ".join(parts) lines[idx + 1] = f"{moved_word} {lines[idx + 1]}" return lines[:max_lines] def _compact_caption_text(text, max_words=None): """Chuẩn hóa phụ đề tiếng Việt, giữ nguyên vẹn 100% nội dung câu.""" text = str(text or "").strip() text = re.sub(r"\s+", " ", text) text = text.strip("\"'“”„”") return text def _theme_colors(theme): theme = str(theme or "motion_drip").lower() if theme == "clean_white": return {"primary": "&H00FFFFFF", "secondary": "&H00FFFFFF"} if theme == "brand_drip": return {"primary": "&H00F2F2F2", "secondary": "&H00D7F7FF"} if theme == "street_luxe": return {"primary": "&H00E8F5FF", "secondary": "&H0099D6FF"} return {"primary": "&H00F7F0DC", "secondary": "&H00A8F4FF"} def _write_modern_ass_from_srt(srt_path, ass_path, width, height, caption_cfg, blur_region=None): events = _parse_srt_events(srt_path) if not events: return None # TikTok-optimized typography: larger, bolder, high contrast for mobile # Vertical video needs bigger font; horizontal keeps moderate size font_name = str(caption_cfg.get("font_name", "Arial Black, Montserrat, Be Vietnam Pro, Arial")) # Increase default for TikTok readability: 82 vertical / 54 horizontal (was 68/48) base_font_size = int(caption_cfg.get("font_size", 82 if height > width else 54)) max_chars = int(caption_cfg.get("max_chars_per_line", 22)) max_lines = int(caption_cfg.get("max_lines", 3)) # TikTok compression demands stronger outline/shadow for legibility outline = float(caption_cfg.get("outline", 4.5)) shadow = float(caption_cfg.get("shadow", 2.2)) shadow_color = "&H99000000" alignment_tag = r"\an8" align_code = 8 font_size = base_font_size if blur_region: try: parts = [int(p) for p in str(blur_region).split(',')] if len(parts) == 6: rx, ry, rw, rh, orig_w, orig_h = parts scale_x = width / float(orig_w) if orig_w > 0 else 1.0 scale_y = height / float(orig_h) if orig_h > 0 else 1.0 box_x = rx * scale_x box_y = ry * scale_y box_w = rw * scale_x box_h = rh * scale_y x = int(box_x + box_w / 2.0) y = int(box_y + box_h / 2.0) alignment_tag = r"\an5" align_code = 5 box_width_pct = max(35.0, min(98.0, (box_w / float(width)) * 100.0)) max_fitted_font = max(24, int(box_h * 0.70)) if font_size > max_fitted_font: font_size = max_fitted_font print(f"[RENDER] Auto-positioned Vietnamese subtitles directly inside blur region: pos=({x},{y}), box_w={box_w:.0f}, font_size={font_size}") else: x = int(width * float(caption_cfg.get("x_percent", 50)) / 100.0) y = int(height * float(caption_cfg.get("y_percent", 78)) / 100.0) box_width_pct = max(45.0, min(96.0, float(caption_cfg.get("box_width_percent", 86)))) except Exception as e: print(f"[RENDER] Warning parsing blur_region: {e}") x = int(width * float(caption_cfg.get("x_percent", 50)) / 100.0) y = int(height * float(caption_cfg.get("y_percent", 78)) / 100.0) box_width_pct = max(45.0, min(96.0, float(caption_cfg.get("box_width_percent", 86)))) else: x = int(width * float(caption_cfg.get("x_percent", 50)) / 100.0) y = int(height * float(caption_cfg.get("y_percent", 78)) / 100.0) box_width_pct = max(45.0, min(96.0, float(caption_cfg.get("box_width_percent", 86)))) # TikTok safe zone: ensure margins keep text inside 86% width / 78% height safe area # Bottom UI of TikTok covers ~ 12% height, right side buttons ~ 10% width → add safe padding safe_margin_v = max(38, int(height * 0.07)) # at least 7% from bottom/top safe_margin_h = max(28, int(width * 0.04)) margin = max(safe_margin_h, int((width * (100.0 - box_width_pct) / 100.0) / 2)) # Clamp y to stay within safe zone (avoid bottom 12% TikTok caption area) # y is calculated above; enforce ceiling at 82% of height for TikTok if 'y' in locals(): max_safe_y = int(height * 0.82) min_safe_y = int(height * 0.12) y = max(min_safe_y, min(max_safe_y, y)) else: # fallback safe y if not yet defined y = int(height * 0.78) uppercase = bool(caption_cfg.get("uppercase", False)) word_jump = bool(caption_cfg.get("word_jump", False)) colors = _theme_colors(caption_cfg.get("theme", "clean_white")) dialogue = [] for index, (start, end, text) in enumerate(events): cleaned_text = _compact_caption_text(text) if uppercase: cleaned_text = cleaned_text.upper() lines = _wrap_caption_words(cleaned_text, max_chars=max_chars, max_lines=max_lines) if not lines: continue cur_font_size = font_size total_len = len(cleaned_text) if len(lines) >= 3 or total_len > 45: cur_font_size = max(22, int(font_size * 0.85)) elif total_len > 30: cur_font_size = max(24, int(font_size * 0.92)) line_prefix = rf"{{{alignment_tag}\pos({x},{y})\fs{cur_font_size}\c{colors['primary']}\3c&H00000000&\bord{outline}\shad{shadow}" if word_jump: line_prefix += r"\fscx98\fscy98\t(0,95,\fscx110\fscy110)\t(95,230,\fscx100\fscy100)" line_prefix += "}" body = line_prefix + r"\N".join(_ass_escape(line) for line in lines) dialogue.append( f"Dialogue: 0,{_ass_time_from_srt(start)},{_ass_time_from_srt(end)},Caption,,{margin},{margin},0,,{body}" ) if not dialogue: return None doc = f"""[Script Info] ScriptType: v4.00+ PlayResX: {width} PlayResY: {height} ScaledBorderAndShadow: yes WrapStyle: 2 [V4+ Styles] Format: Name, Fontname, Fontsize, PrimaryColour, SecondaryColour, OutlineColour, BackColour, Bold, Italic, Underline, StrikeOut, ScaleX, ScaleY, Spacing, Angle, BorderStyle, Outline, Shadow, Alignment, MarginL, MarginR, MarginV, Encoding Style: Caption,{font_name},{font_size},{colors['primary']},{colors['secondary']},&H00000000,{shadow_color},-1,0,0,0,100,100,0.8,0,1,{outline},{shadow},{align_code},{margin},{margin},0,1 [Events] Format: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text {chr(10).join(dialogue)} """ ass_path = Path(ass_path) ass_path.parent.mkdir(parents=True, exist_ok=True) ass_path.write_text(doc, encoding="utf-8-sig") return ass_path def _region_to_native(blur_region, vid_w, vid_h): """Parse 'rx,ry,rw,rh,orig_w,orig_h' and re-derive the real blur/crop coords against the ACTUAL probed video dimensions using normalized fractions.""" parts = [int(p) for p in str(blur_region).split(',')] if len(parts) != 6: return None rx, ry, rw, rh, reg_orig_w, reg_orig_h = parts norm_x = rx / float(reg_orig_w) if reg_orig_w > 0 else 0.0 norm_y = ry / float(reg_orig_h) if reg_orig_h > 0 else 0.0 norm_w = rw / float(reg_orig_w) if reg_orig_w > 0 else 0.0 norm_h = rh / float(reg_orig_h) if reg_orig_h > 0 else 0.0 real_x = int(round(norm_x * vid_w)) real_y = int(round(norm_y * vid_h)) real_w = max(1, int(round(norm_w * vid_w))) real_h = max(1, int(round(norm_h * vid_h))) real_w = min(real_w, vid_w) real_h = min(real_h, vid_h) real_x = max(0, min(real_x, vid_w - real_w)) real_y = max(0, min(real_y, vid_h - real_h)) print(f"[RENDER] [Normalized Cords] x={norm_x:.4f}, y={norm_y:.4f}, w={norm_w:.4f}, h={norm_h:.4f}") print(f"[RENDER] [Native Frame Cords] x={real_x}, y={real_y}, w={real_w}, h={real_h} (frame {vid_w}x{vid_h})") return real_x, real_y, real_w, real_h, vid_w, vid_h def main(): parser = argparse.ArgumentParser(description="Standalone Video Render & Merge Worker CLI") parser.add_argument("--video", required=True, help="Path to input video file") parser.add_argument("--audio", required=True, help="Path to input mixed audio WAV file") parser.add_argument("--output", required=True, help="Path to output video file") parser.add_argument("--srt", default="", help="Path to translated SRT file for hardsubs") parser.add_argument("--blur-region", default="", help="Blur region as 'rx,ry,rw,rh,orig_w,orig_h'") parser.add_argument("--subtitle-region-mode", default="delogo", choices=["delogo", "blur", "none"], help="How to cover the selected source subtitle region") parser.add_argument("--preserve-regions", default="", help="Path to preserve_regions.json") parser.add_argument("--ffmpeg-path", default="ffmpeg", help="Path to ffmpeg executable") parser.add_argument("--encoder", default="auto", choices=["auto", "nvenc", "h264_nvenc", "libx264"], help="Video encoder to use") parser.add_argument("--allow-cpu-fallback", default="true", help="Allow fallback to CPU (true/false)") parser.add_argument("--zoom", default="1.0", help="Auto-zoom factor (e.g. 1.15 = zoom in 115%%), 1.0 disables") parser.add_argument("--trim-head-silence", default="0", help="Trim leading silence of video+audio by N seconds (0 disables)") parser.add_argument("--logo", default="", help="Path to logo image for overlay (PNG/JPG)") parser.add_argument("--logo-preset", default="bottom_right", help="Logo position preset") parser.add_argument("--logo-scale", default="15", help="Logo scale as %% of video width (5-40)") parser.add_argument("--logo-opacity", default="1.0", help="Logo opacity 0.0-1.0") parser.add_argument("--logo-x-pct", default="80", help="Logo custom X %%") parser.add_argument("--logo-y-pct", default="80", help="Logo custom Y %%") parser.add_argument("--logo-chromakey", default="true", help="Enable green chromakey removal (true/false)") parser.add_argument("--cta", default="", help="Path to CTA video for overlay (MP4/MOV with green bg)") parser.add_argument("--cta-preset", default="bottom_center", help="CTA position preset") parser.add_argument("--cta-scale", default="48", help="CTA scale as %% of video width (15-60) - TikTok-optimized default 48") parser.add_argument("--cta-opacity", default="1.0", help="CTA opacity 0.0-1.0") parser.add_argument("--cta-x-pct", default="50", help="CTA custom X %%") parser.add_argument("--cta-y-pct", default="85", help="CTA custom Y %%") parser.add_argument("--cta-interval", default="5.0", help="CTA appear interval seconds") parser.add_argument("--cta-duration", default="2.0", help="CTA visible duration per interval seconds") parser.add_argument("--cta-chromakey", default="true", help="Enable green chromakey for CTA (true/false)") args = parser.parse_args() video_path = Path(args.video) audio_path = Path(args.audio) output_path = Path(args.output) ffmpeg_path = Path(args.ffmpeg_path) if not video_path.exists(): print(f"Error: Input video not found at {video_path}", file=sys.stderr) sys.exit(1) if not audio_path.exists(): print(f"Error: Input audio not found at {audio_path}", file=sys.stderr) sys.exit(1) # 1. Parse preserve regions preserve_intervals = [] if args.preserve_regions: pr_path = Path(args.preserve_regions) if pr_path.exists(): try: with open(pr_path, "r", encoding="utf-8") as f: preserve_raw = json.load(f) # Convert ms to seconds preserve_intervals = [(x[0] / 1000.0, x[1] / 1000.0) for x in preserve_raw] print(f"Loaded preserve intervals: {len(preserve_intervals)} intervals.") except Exception as e: print(f"Warning: Failed to load preserve regions: {e}", file=sys.stderr) # Studio director: auto-zoom + trim leading silence zoom_factor = 1.0 try: zoom_factor = float(args.zoom) except Exception: zoom_factor = 1.0 trim_head_sec = 0.0 try: trim_head_sec = float(args.trim_head_silence) except Exception: trim_head_sec = 0.0 # When trimming head silence we must shift the SRT timeline BEFORE generating ASS. if trim_head_sec > 0 and args.srt: try: src_srt = Path(args.srt) shifted_srt = src_srt.with_name(src_srt.stem + "_shifted.srt") import shutil as _sh _sh.copy2(str(src_srt), str(shifted_srt)) _shift_srt(shifted_srt, trim_head_sec) args.srt = str(shifted_srt) print(f"[RENDER] Shifted subtitle timeline by -{trim_head_sec:.2f}s after head-silence trim.") except Exception as e: print(f"[RENDER] Warning: could not shift srt: {e}", file=sys.stderr) # ── Logo & CTA detection ── logo_enabled = bool(args.logo and Path(args.logo).exists()) logo_scale = 15.0 logo_opacity = 1.0 logo_preset = "bottom_right" logo_x_pct = 80.0 logo_y_pct = 80.0 logo_chromakey = True if logo_enabled: try: logo_scale = float(args.logo_scale) logo_opacity = float(args.logo_opacity) logo_preset = str(args.logo_preset) logo_x_pct = float(args.logo_x_pct) logo_y_pct = float(args.logo_y_pct) logo_chromakey = str(args.logo_chromakey).lower() in ("true","1","yes","t") print(f"[RENDER] Logo overlay: {Path(args.logo).name} preset={logo_preset} scale={logo_scale}% opacity={logo_opacity} chromakey={logo_chromakey}") except Exception as _e: print(f"[RENDER] Logo param parse error: {_e}", file=sys.stderr) cta_enabled = bool(args.cta and Path(args.cta).exists()) cta_scale = 35.0 cta_opacity = 1.0 cta_preset = "bottom_center" cta_x_pct = 50.0 cta_y_pct = 85.0 cta_interval = 5.0 cta_duration = 5.0 cta_chromakey = True if cta_enabled: try: cta_scale = float(args.cta_scale) cta_opacity = float(args.cta_opacity) cta_preset = str(args.cta_preset) cta_x_pct = float(args.cta_x_pct) cta_y_pct = float(args.cta_y_pct) cta_interval = float(args.cta_interval) cta_duration = float(args.cta_duration) cta_chromakey = str(args.cta_chromakey).lower() in ("true","1","yes","t") print(f"[RENDER] CTA overlay: {Path(args.cta).name} preset={cta_preset} scale={cta_scale}% every {cta_interval}s chromakey={cta_chromakey}") except Exception as _e: print(f"[RENDER] CTA param parse error: {_e}", file=sys.stderr) # 2. Build FFmpeg filter complex if blur or hardsub or logo or cta is needed filter_complex = None if args.blur_region or args.srt or zoom_factor > 1.0 or logo_enabled or cta_enabled: try: # Get video dimensions - try cv2 then ffprobe fallback for correct vertical/horizontal handling orig_w, orig_h = 1920, 1080 if args.video: try: import cv2 cap = cv2.VideoCapture(str(video_path)) w = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH)) or 0 h = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT)) or 0 cap.release() if w > 0 and h > 0: orig_w, orig_h = w, h else: raise ValueError("cv2 returned 0") except Exception: # ffprobe fallback (handles vertical videos correctly when cv2 unavailable) try: import subprocess as _sp, json as _js, shutil as _sh _ffprobe = _sh.which("ffprobe") or str(Path(args.ffmpeg_path).parent / "ffprobe.exe") if args.ffmpeg_path else "ffprobe" if not Path(_ffprobe).exists(): _ffprobe = "ffprobe" _res = _sp.run([_ffprobe, "-v", "error", "-select_streams", "v:0", "-show_entries", "stream=width,height", "-of", "json", str(video_path)], capture_output=True, text=True, timeout=5) _j = _js.loads(_res.stdout or "{}") _ws = _j.get("streams", [{}])[0] if _ws.get("width") and _ws.get("height"): orig_w, orig_h = int(_ws["width"]), int(_ws["height"]) except Exception: pass render_cfg = {} try: config_path = Path(__file__).parent.parent.parent / "config.json" if config_path.exists(): with open(config_path, "r", encoding="utf-8") as f: cfg_data = json.load(f) render_cfg = cfg_data.get("render", {}) except Exception as e: print(f"Warning: Failed to load render config: {e}", file=sys.stderr) dynamic_fontsize = 16 ass_playres_y = 288 ass_playres_x = int(288 * orig_w / orig_h) if orig_h else 384 if args.blur_region: converted = _region_to_native(args.blur_region, orig_w, orig_h) if converted is None: print("Error: invalid blur_region format, expected 'rx,ry,rw,rh,orig_w,orig_h'", file=sys.stderr) sys.exit(1) rx, ry, rw, rh, orig_w, orig_h = converted # TikTok-safe delogo inset: ensure 2px border inside frame to avoid "outside of frame" error # and guarantee even dimensions for yuv420p rx = max(2, rx) ry = max(2, ry) rw = min(rw, orig_w - rx - 2) rh = min(rh, orig_h - ry - 2) rx = rx if rx % 2 == 0 else max(0, rx - 1) ry = ry if ry % 2 == 0 else max(0, ry - 1) rw = rw if rw % 2 == 0 else rw - 1 rh = rh if rh % 2 == 0 else rh - 1 rw, rh = max(4, rw), max(4, rh) scale_x = ass_playres_x / orig_w if orig_w else 1 scale_y = ass_playres_y / orig_h if orig_h else 1 ass_rx = int(rx * scale_x) ass_ry = int(ry * scale_y) ass_rw = int(rw * scale_x) ass_rh = int(rh * scale_y) ass_margin_l = ass_rx ass_margin_r = max(0, ass_playres_x - (ass_rx + ass_rw)) disable_blur_on_preserve_regions = render_cfg.get("disable_blur_on_preserve_regions", True) if preserve_intervals and disable_blur_on_preserve_regions: ass_margin_v = 12 else: sub_block_h = int(2 * 1.35 * dynamic_fontsize) sub_block_h = min(sub_block_h, ass_rh) ass_bottom_of_text = ass_ry + ass_rh // 2 + sub_block_h // 2 ass_margin_v = max(0, ass_playres_y - ass_bottom_of_text) force_style = ( f"FontName=Arial,FontSize={dynamic_fontsize}," f"PrimaryColour=&H00FFFFFF,OutlineColour=&H00000000,BackColour=&H00000000," f"BorderStyle=1,Outline=2,Shadow=1,Alignment=2," f"MarginL={ass_margin_l},MarginR={ass_margin_r},MarginV={ass_margin_v}," f"WrapStyle=1" ) else: force_style = ( f"FontName=Arial,FontSize={dynamic_fontsize}," f"PrimaryColour=&H00FFFFFF,OutlineColour=&H00000000,BackColour=&H00000000," f"BorderStyle=1,Outline=2,Shadow=1,Alignment=2," f"MarginV=15,WrapStyle=1" ) # Build video blur / inpaint filter on original resolution video_filter = None if args.blur_region and args.subtitle_region_mode != "none": disable_blur_on_preserve_regions = render_cfg.get("disable_blur_on_preserve_regions", True) if preserve_intervals and disable_blur_on_preserve_regions: top_h = int(rh * 0.5) bot_y = ry + top_h bot_h = rh - top_h enable_str = "+".join( [f"between(t,{s:.3f},{e:.3f})" for s, e in preserve_intervals] ) if args.subtitle_region_mode == "blur": video_filter = ( f"[0:v]split[base][crop];" f"[crop]crop={rw}:{top_h}:{rx}:{ry},gblur=sigma=18[topblur];" f"[base][topblur]overlay={rx}:{ry}[v1];" f"[v1]split[base2][crop2];" f"[crop2]crop={rw}:{bot_h}:{rx}:{bot_y},gblur=sigma=18[botblur];" f"[base2][botblur]overlay={rx}:{bot_y}:enable='not({enable_str})'[blurred]" ) else: video_filter = ( f"[0:v]delogo=x={rx}:y={ry}:w={rw}:h={top_h}[v1];" f"[v1]delogo=x={rx}:y={bot_y}:w={rw}:h={bot_h}:enable='not({enable_str})'[blurred]" ) else: if args.subtitle_region_mode == "blur": video_filter = ( f"[0:v]split[base][crop];" f"[crop]crop={rw}:{rh}:{rx}:{ry},gblur=sigma=20[regionblur];" f"[base][regionblur]overlay={rx}:{ry}[blurred]" ) else: video_filter = f"[0:v]delogo=x={rx}:y={ry}:w={rw}:h={rh}[blurred]" # Build subtitle filter srt_filter = None if args.srt: caption_cfg = render_cfg.get("caption", {}) if isinstance(render_cfg, dict) else {} use_modern_ass = bool(caption_cfg.get("enabled", True)) ass_path = None if use_modern_ass: ass_blur_region = ( f"{rx},{ry},{rw},{rh},{orig_w},{orig_h}" if args.blur_region else None ) ass_path = _write_modern_ass_from_srt( args.srt, output_path.parent / "translated_caption.ass", orig_w, orig_h, caption_cfg, blur_region=ass_blur_region, ) if ass_path: ass_abs = str(Path(ass_path).resolve()).replace("\\", "/").replace(":", "\\:") srt_filter = f"subtitles='{ass_abs}'" print(f"[RENDER] modern ASS captions enabled: {ass_path}") else: srt_abs = str(Path(args.srt).resolve()).replace("\\", "/").replace(":", "\\:") srt_filter = f"subtitles='{srt_abs}':force_style='{force_style}'" # Auto-zoom filter string zoom_filter = None if zoom_factor > 1.0: zoom_filter = ( f"scale={int(orig_w*zoom_factor)}:{int(orig_h*zoom_factor)}:flags=lanczos," f"crop={orig_w}:{orig_h}:((iw-ow)/2):((ih-oh)/2),setsar=1" ) print(f"[RENDER] auto-zoom {zoom_factor}x enabled") # Combine filters: blur -> zoom -> logo -> subtitles # Build step-by-step with intermediate labels filter_parts = [] cur_label = None # Video blur/delogo (outputs [blurred] if exists) if video_filter: filter_parts.append(video_filter) cur_label = "blurred" else: cur_label = "0:v" # Zoom if zoom_filter: next_label = "v" # Reserve intermediate if logo/cta or subtitles will follow if logo_enabled or cta_enabled or srt_filter: next_label = "zm" filter_parts.append(f"[{cur_label}]{zoom_filter}[{next_label}]") cur_label = next_label # Logo overlay (logo is input 2) if logo_enabled: # Probe logo size for scale _lw, _lh = 2400, 1792 try: _img = cv2.imread(str(Path(args.logo)), cv2.IMREAD_UNCHANGED) if _img is not None: _lh, _lw = _img.shape[:2] except Exception: pass _target_w = orig_w * logo_scale / 100.0 _sf = max(0.02, min(0.5, _target_w / max(1, _lw))) # Clamp height to 85% of video height _target_h = _lh * _sf _max_h = orig_h * 0.85 if _target_h > _max_h: _sf = _max_h / max(1, _lh) _logo_vf_parts = [] if logo_chromakey: _logo_vf_parts.append("colorkey=0x00FF00:0.3:0.1") _logo_vf_parts.append("format=rgba") _logo_vf_parts.append(f"scale=iw*{_sf:.4f}:ih*{_sf:.4f}:flags=lanczos") if logo_opacity < 0.99: _logo_vf_parts.append(f"colorchannelmixer=aa={logo_opacity:.2f}") _logo_vf = ",".join(_logo_vf_parts) # Position _m = 2.0 if logo_preset == "top_left": _x, _y = f"W*{_m/100:.3f}", f"H*{_m/100:.3f}" elif logo_preset == "top_right": _x, _y = f"W-w-W*{_m/100:.3f}", f"H*{_m/100:.3f}" elif logo_preset == "bottom_left": _x, _y = f"W*{_m/100:.3f}", f"H-h-H*{_m/100:.3f}" elif logo_preset == "bottom_right": _x, _y = f"W-w-W*{_m/100:.3f}", f"H-h-H*{_m/100:.3f}" elif logo_preset == "center": _x, _y = "(W-w)/2", "(H-h)/2" elif logo_preset == "top_center": _x, _y = "(W-w)/2", f"H*{_m/100:.3f}" elif logo_preset == "bottom_center": _x, _y = "(W-w)/2", f"H-h-H*{_m/100:.3f}" elif logo_preset == "custom": _x, _y = f"W*{logo_x_pct/100:.4f}", f"H*{logo_y_pct/100:.4f}" else: _x, _y = f"W-w-W*{_m/100:.3f}", f"H-h-H*{_m/100:.3f}" filter_parts.append(f"[2:v]{_logo_vf}[logo]") next_label = "v" if not (cta_enabled or srt_filter) else "with_logo" filter_parts.append(f"[{cur_label}][logo]overlay={_x}:{_y}:format=rgb[{next_label}]") cur_label = next_label # CTA video overlay (periodic every interval, e.g., 5s) if cta_enabled: _cta_idx = 3 if logo_enabled else 2 # Probe CTA size _cw, _ch = 1080, 1920 try: _cap = cv2.VideoCapture(str(Path(args.cta))) _cw = int(_cap.get(cv2.CAP_PROP_FRAME_WIDTH)) or _cw _ch = int(_cap.get(cv2.CAP_PROP_FRAME_HEIGHT)) or _ch _cap.release() except Exception: pass # TikTok CTA bump: enforce minimum 42% width so CTA is clearly visible on mobile effective_cta_scale = max(float(cta_scale), 42.0) _cta_target_w = orig_w * effective_cta_scale / 100.0 _cta_sf = max(0.05, min(1.0, _cta_target_w / max(1, _cw))) _cta_target_h = _ch * _cta_sf _max_h2 = orig_h * 0.90 if _cta_target_h > _max_h2: _cta_sf = _max_h2 / max(1, _ch) _cta_vf_parts = [] if cta_chromakey: _cta_vf_parts.append("colorkey=0x00FF00:0.3:0.1") _cta_vf_parts.append("format=rgba") _cta_vf_parts.append(f"scale=iw*{_cta_sf:.4f}:ih*{_cta_sf:.4f}:flags=lanczos") if cta_opacity < 0.99: _cta_vf_parts.append(f"colorchannelmixer=aa={cta_opacity:.2f}") _cta_vf = ",".join(_cta_vf_parts) _m2 = 2.0 if cta_preset == "top_left": _cx, _cy = f"W*{_m2/100:.3f}", f"H*{_m2/100:.3f}" elif cta_preset == "top_right": _cx, _cy = f"W-w-W*{_m2/100:.3f}", f"H*{_m2/100:.3f}" elif cta_preset == "bottom_left": _cx, _cy = f"W*{_m2/100:.3f}", f"H-h-H*{_m2/100:.3f}" elif cta_preset == "bottom_right": _cx, _cy = f"W-w-W*{_m2/100:.3f}", f"H-h-H*{_m2/100:.3f}" elif cta_preset == "center": _cx, _cy = "(W-w)/2", "(H-h)/2" elif cta_preset == "top_center": _cx, _cy = "(W-w)/2", f"H*{_m2/100:.3f}" elif cta_preset == "bottom_center": _cx, _cy = "(W-w)/2", f"H-h-H*{_m2/100:.3f}" elif cta_preset == "custom": _cx, _cy = f"W*{cta_x_pct/100:.4f}", f"H*{cta_y_pct/100:.4f}" else: _cx, _cy = "(W-w)/2", f"H-h-H*{_m2/100:.3f}" _enable = f"lt(mod(t\\,{cta_interval})\\,{cta_duration})" filter_parts.append(f"[{_cta_idx}:v]{_cta_vf}[cta]") next_label2 = "v" if not srt_filter else "with_cta" filter_parts.append(f"[{cur_label}][cta]overlay={_cx}:{_cy}:format=rgb:enable='{_enable}'[{next_label2}]") cur_label = next_label2 # ── TikTok polish: CFR 30fps + even dims + lanczos + light sharpen (before subs to keep text razor sharp) # This ensures sharp detail after blur/delogo/overlay without oversharpen tiktok_polish = "fps=30:round=near,scale=trunc(iw/2)*2:trunc(ih/2)*2:flags=lanczos+accurate_rnd+full_chroma_int:sws_dither=ed,unsharp=3:3:0.35:3:3:0.0" # Inject polish before subtitles (keep subtitles as final top layer) if srt_filter: # Need intermediate polish step if cur_label == "v": filter_parts.append(f"[{cur_label}]{tiktok_polish}[polished];[polished]{srt_filter}[v]") else: filter_parts.append(f"[{cur_label}]{tiktok_polish}[polished]") cur_label = "polished" filter_parts.append(f"[{cur_label}]{srt_filter}[v]") cur_label = "v" else: # No subtitles but still need TikTok CFR/sharpen/even polish for upload compliance if cur_label != "0:v": # Only add polish if we already have processing chain (avoid redundant filter on copy path) filter_parts.append(f"[{cur_label}]{tiktok_polish}[v]") cur_label = "v" # If filter_parts empty but logo only (no blur/zoom/srt) we already handled logo # If still no filter (should not happen), set to null if filter_parts: filter_complex = ";".join(filter_parts) print(f"[RENDER] filter complex pipeline: {filter_complex}") else: filter_complex = None except Exception as e: print(f"Error building filter complex: {e}", file=sys.stderr) sys.exit(1) allow_cpu_fallback = args.allow_cpu_fallback.lower() in ("true", "1", "yes", "t") # NVENC preflight check is_nvenc_requested = args.encoder in ("auto", "nvenc", "h264_nvenc") if is_nvenc_requested: print("Running NVENC preflight check...") if not check_nvenc_available(ffmpeg_path): print("NVENC preflight check failed.", file=sys.stderr) if not allow_cpu_fallback: print("NVENC unavailable or driver/API mismatch.\n" "CPU fallback disabled by strict GPU policy.\n" "Suggested fix: update NVIDIA driver or use FFmpeg build compatible with current driver.", file=sys.stderr) sys.exit(3) else: print("Warning: NVENC preflight check failed. Driver/API mismatch. CPU fallback enabled, falling back to libx264.") args.encoder = "libx264" # Resolve selected encoder for logging purposes selected_encoder = "h264_nvenc" if args.encoder in ("nvenc", "h264_nvenc") else (args.encoder if args.encoder != "auto" else "h264_nvenc") print(f"[RENDER] selected encoder: {selected_encoder}") print(f"[RENDER] cpu fallback: {'true' if allow_cpu_fallback else 'false'}") # 3. Determine encoders to try encoders_to_try = [] if args.encoder in ("nvenc", "h264_nvenc"): encoders_to_try = ["h264_nvenc"] if allow_cpu_fallback: encoders_to_try.append("libx264") elif args.encoder == "libx264": encoders_to_try = ["libx264"] else: # auto encoders_to_try = ["h264_nvenc"] if allow_cpu_fallback: encoders_to_try.append("libx264") output_path.parent.mkdir(parents=True, exist_ok=True) success = False for encoder in encoders_to_try: print(f"Attempting to render video using encoder '{encoder}'...") # TikTok spec: CFR, genpts, accurate seek, 48k AAC cmd = [ str(ffmpeg_path), "-y", "-noautorotate", "-fflags", "+genpts", "-avoid_negative_ts", "make_zero", ] if trim_head_sec > 0: cmd.extend(["-ss", f"{trim_head_sec:.3f}"]) cmd.extend(["-i", str(video_path)]) if trim_head_sec > 0: cmd.extend(["-ss", f"{trim_head_sec:.3f}"]) cmd.extend(["-i", str(audio_path)]) # Logo & CTA extra inputs (must be after video+audio so filter can ref [2:v]/[3:v]) _logo_enabled_cmd = bool(getattr(args, 'logo', '') and Path(args.logo).exists()) if _logo_enabled_cmd: cmd.extend(["-loop", "1", "-i", str(Path(args.logo))]) _cta_enabled_cmd = bool(getattr(args, 'cta', '') and Path(args.cta).exists()) if _cta_enabled_cmd: cmd.extend(["-stream_loop", "999", "-i", str(Path(args.cta))]) # ── TikTok output compliance: ensure filter even if no blur/subs (CFR + sharpen) active_filter = filter_complex if not active_filter: # Minimal TikTok polish when no other processing: CFR 30 + even dims + light sharpen active_filter = "[0:v]fps=30:round=near,scale=trunc(iw/2)*2:trunc(ih/2)*2:flags=lanczos+accurate_rnd+full_chroma_int:sws_dither=ed,unsharp=3:3:0.35:3:3:0.0[v]" # Add -shortest when looped logo/CTA present to avoid infinite encode _need_shortest = logo_enabled or cta_enabled if _need_shortest: cmd.extend(["-filter_complex", active_filter, "-map", "[v]", "-map", "1:a:0", "-shortest"]) else: cmd.extend(["-filter_complex", active_filter, "-map", "[v]", "-map", "1:a:0"]) # ── Video encoder: H.264 High Profile 8-bit yuv420p, CFR 30, high detail if encoder == "h264_nvenc": cmd.extend([ "-r", "30", "-c:v", "h264_nvenc", "-preset", "p4", "-tune", "hq", "-profile:v", "high", "-level", "4.1", "-rc", "vbr", "-cq", "19", "-b:v", "0", "-maxrate", "8M", "-bufsize", "12M", "-g", "60", "-keyint_min", "30", "-sc_threshold", "0", "-pix_fmt", "yuv420p", "-colorspace", "bt709", "-color_primaries", "bt709", "-color_trc", "bt709", "-color_range", "tv", ]) else: cmd.extend([ "-r", "30", "-c:v", "libx264", "-preset", "medium", "-crf", "18", "-profile:v", "high", "-level", "4.1", "-pix_fmt", "yuv420p", "-g", "60", "-keyint_min", "30", "-sc_threshold", "0", "-x264-params", "ref=4:bframes=2:me=hex:subme=7:psy=1:psy-rd=0.8:aq-mode=2", "-colorspace", "bt709", "-color_primaries", "bt709", "-color_trc", "bt709", "-color_range", "tv", ]) # ── Audio: AAC-LC 48kHz 192k, 2ch, precise sync cmd.extend([ "-c:a", "aac", "-profile:a", "aac_low", "-ar", "48000", "-ac", "2", "-b:a", "192k", "-af", "aresample=async=1:min_hard_comp=0.100000:first_pts=0", ]) # ── TikTok faststart + interleaving for AV sync & timestamp correctness cmd.extend([ "-movflags", "+faststart", "-fflags", "+genpts", "-max_interleave_delta", "100M", "-vsync", "cfr", "-fps_mode", "cfr", "-shortest", ]) cmd.append(str(output_path)) print(f"Executing: {' '.join(cmd)}") startupinfo = None if sys.platform == 'win32': startupinfo = subprocess.STARTUPINFO() startupinfo.dwFlags |= subprocess.STARTF_USESHOWWINDOW try: res = subprocess.run( cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, encoding="utf-8", errors="ignore", startupinfo=startupinfo, timeout=1200 ) if res.returncode == 0: print(f"Render completed successfully using encoder '{encoder}'.") success = True break else: print(f"Warning: Encoder '{encoder}' failed with exit code {res.returncode}.", file=sys.stderr) print(f"FFmpeg Stderr:\n{res.stderr}", file=sys.stderr) except Exception as e: print(f"Warning: Exception using encoder '{encoder}': {e}", file=sys.stderr) if not success: if not allow_cpu_fallback and (args.encoder in ("nvenc", "h264_nvenc") or args.encoder == "auto"): print("NVENC requested but unavailable. CPU render fallback disabled by strict GPU policy.", file=sys.stderr) sys.exit(3) print("Error: All rendering encoders failed.", file=sys.stderr) sys.exit(2) sys.exit(0) if __name__ == "__main__": main()