Spaces:
Running on Zero
Running on Zero
Download app/core/render_worker_cli.py from hoangtaiii/DRIPPY4: direct link, hf CLI and curl.
- Browser
- Download file 44.3 kB
-
https://huggingface.co/spaces/hoangtaiii/DRIPPY4/resolve/main/app/core/render_worker_cli.py
- Command line
-
hf download hf://spaces/hoangtaiii/DRIPPY4/app/core/render_worker_cli.py
-
curl -L -o render_worker_cli.py https://huggingface.co/spaces/hoangtaiii/DRIPPY4/resolve/main/app/core/render_worker_cli.py
44.3 kB
| import sys | |
| import os | |
| import re | |
| import json | |
| import argparse | |
| import subprocess | |
| from pathlib import Path | |
| # Enforce UTF-8 for Windows console | |
| if sys.platform == 'win32': | |
| try: | |
| if hasattr(sys.stdout, 'reconfigure'): | |
| sys.stdout.reconfigure(encoding='utf-8') | |
| if hasattr(sys.stderr, 'reconfigure'): | |
| sys.stderr.reconfigure(encoding='utf-8') | |
| except Exception: | |
| pass | |
| def check_nvenc_available(ffmpeg_path): | |
| cmd = [ | |
| str(ffmpeg_path), "-y", | |
| "-f", "lavfi", "-i", "color=c=black:s=256x256", | |
| "-t", "1", | |
| "-c:v", "h264_nvenc", | |
| "-f", "null", "-" | |
| ] | |
| try: | |
| startupinfo = None | |
| if sys.platform == 'win32': | |
| startupinfo = subprocess.STARTUPINFO() | |
| startupinfo.dwFlags |= subprocess.STARTF_USESHOWWINDOW | |
| res = subprocess.run( | |
| cmd, | |
| stdout=subprocess.PIPE, | |
| stderr=subprocess.PIPE, | |
| text=True, | |
| encoding="utf-8", | |
| errors="ignore", | |
| startupinfo=startupinfo, | |
| timeout=10 | |
| ) | |
| return res.returncode == 0 | |
| except Exception: | |
| return False | |
| def _ass_time_from_srt(value): | |
| m = re.match(r"^(\d{2}):(\d{2}):(\d{2})[,.](\d{3})$", str(value).strip()) | |
| if not m: | |
| return "0:00:00.00" | |
| hh, mm, ss, ms = [int(x) for x in m.groups()] | |
| centis = int(round(ms / 10.0)) | |
| return f"{hh}:{mm:02d}:{ss:02d}.{centis:02d}" | |
| def _parse_srt_events(srt_path): | |
| try: | |
| content = Path(srt_path).read_text(encoding="utf-8", errors="ignore").strip().replace("\r\n", "\n") | |
| except Exception: | |
| return [] | |
| pattern = r"(\d+)\s+(\d{2}:\d{2}:\d{2}[.,]\d{3})\s*-->\s*(\d{2}:\d{2}:\d{2}[.,]\d{3})\s*\n(.*?)(?=\n\s*\d+\s+\d{2}:\d{2}:\d{2}[.,]\d{3}\s*-->|\Z)" | |
| events = [] | |
| for match in re.finditer(pattern, content, re.DOTALL): | |
| text = " ".join(line.strip() for line in match.group(4).splitlines() if line.strip()) | |
| if text: | |
| events.append((match.group(2), match.group(3), text)) | |
| return events | |
| def _ass_escape(text): | |
| return str(text or "").replace("\\", "\\\\").replace("{", r"\{").replace("}", r"\}").replace("\n", r"\N") | |
| def _shift_srt(srt_path, seconds): | |
| """Shift every timestamp in an SRT file by -seconds (in place). Used after | |
| trimming leading silence so hardsub captions stay aligned with the video.""" | |
| srt_path = Path(srt_path) | |
| if seconds <= 0 or not srt_path.exists(): | |
| return | |
| content = srt_path.read_text(encoding="utf-8", errors="ignore").replace("\r\n", "\n") | |
| def shift_ts(m): | |
| def to_ms(t): | |
| tm = re.match(r"(\d+):(\d+):(\d+)[.,](\d+)", t) | |
| if not tm: | |
| return 0 | |
| hh, mm, ss, ms = (int(x) for x in tm.groups()) | |
| return ((hh * 3600 + mm * 60 + ss) * 1000) + ms | |
| def from_ms(v): | |
| v = max(0, int(v)) | |
| hh, rem = divmod(v, 3600000) | |
| mm, rem = divmod(rem, 60000) | |
| ss, ms = divmod(rem, 1000) | |
| return f"{hh:02d}:{mm:02d}:{ss:02d},{ms:03d}" | |
| delta = int(round(seconds * 1000)) | |
| return f"{from_ms(to_ms(m.group(1)) - delta)} --> {from_ms(to_ms(m.group(2)) - delta)}" | |
| pattern = re.compile(r"(\d{2}:\d{2}:\d{2}[.,]\d{3})\s*-->\s*(\d{2}:\d{2}:\d{2}[.,]\d{3})") | |
| new_content = pattern.sub(shift_ts, content) | |
| srt_path.write_text(new_content, encoding="utf-8") | |
| def _wrap_caption_words(text, max_chars=22, max_lines=3): | |
| """Wrap caption text naturally across lines without dropping words.""" | |
| words = str(text or "").split() | |
| if not words: | |
| return [] | |
| lines = [] | |
| current = "" | |
| for word in words: | |
| candidate = word if not current else f"{current} {word}" | |
| if len(candidate) <= max_chars or not current: | |
| current = candidate | |
| else: | |
| if len(lines) < max_lines - 1: | |
| lines.append(current) | |
| current = word | |
| else: | |
| # α» dΓ²ng cuα»i cΓΉng, tiαΊΏp tα»₯c gα»p toΓ n bα» tα»« cΓ²n lαΊ‘i vΓ o dΓ²ng nΓ y, khΓ΄ng Δược bα» sΓ³t | |
| current = f"{current} {word}" | |
| if current: | |
| lines.append(current) | |
| # Xα» lΓ½ cΓ‘c tα»« nα»i khΓ΄ng Δα»©ng lΖ‘ lα»ng mα»t mΓ¬nh α» cuα»i dΓ²ng | |
| connectors = {"vΓ ", "vα»i", "cα»§a", "lΓ ", "thΓ¬", "mΓ ", "rαΊ±ng", "nΓͺn", "nhΖ°ng", "hoαΊ·c", "vΓ¬"} | |
| if len(lines) > 1: | |
| for idx in range(len(lines) - 1): | |
| parts = lines[idx].split() | |
| if len(parts) > 1 and parts[-1].strip(" ,.!?;:").lower() in connectors: | |
| moved_word = parts.pop() | |
| lines[idx] = " ".join(parts) | |
| lines[idx + 1] = f"{moved_word} {lines[idx + 1]}" | |
| return lines[:max_lines] | |
| def _compact_caption_text(text, max_words=None): | |
| """ChuαΊ©n hΓ³a phα»₯ Δα» tiαΊΏng Viα»t, giα»― nguyΓͺn vαΊΉn 100% nα»i dung cΓ’u.""" | |
| text = str(text or "").strip() | |
| text = re.sub(r"\s+", " ", text) | |
| text = text.strip("\"'ββββ") | |
| return text | |
| def _theme_colors(theme): | |
| theme = str(theme or "motion_drip").lower() | |
| if theme == "clean_white": | |
| return {"primary": "&H00FFFFFF", "secondary": "&H00FFFFFF"} | |
| if theme == "brand_drip": | |
| return {"primary": "&H00F2F2F2", "secondary": "&H00D7F7FF"} | |
| if theme == "street_luxe": | |
| return {"primary": "&H00E8F5FF", "secondary": "&H0099D6FF"} | |
| return {"primary": "&H00F7F0DC", "secondary": "&H00A8F4FF"} | |
| def _write_modern_ass_from_srt(srt_path, ass_path, width, height, caption_cfg, blur_region=None): | |
| events = _parse_srt_events(srt_path) | |
| if not events: | |
| return None | |
| # TikTok-optimized typography: larger, bolder, high contrast for mobile | |
| # Vertical video needs bigger font; horizontal keeps moderate size | |
| font_name = str(caption_cfg.get("font_name", "Arial Black, Montserrat, Be Vietnam Pro, Arial")) | |
| # Increase default for TikTok readability: 82 vertical / 54 horizontal (was 68/48) | |
| base_font_size = int(caption_cfg.get("font_size", 82 if height > width else 54)) | |
| max_chars = int(caption_cfg.get("max_chars_per_line", 22)) | |
| max_lines = int(caption_cfg.get("max_lines", 3)) | |
| # TikTok compression demands stronger outline/shadow for legibility | |
| outline = float(caption_cfg.get("outline", 4.5)) | |
| shadow = float(caption_cfg.get("shadow", 2.2)) | |
| shadow_color = "&H99000000" | |
| alignment_tag = r"\an8" | |
| align_code = 8 | |
| font_size = base_font_size | |
| if blur_region: | |
| try: | |
| parts = [int(p) for p in str(blur_region).split(',')] | |
| if len(parts) == 6: | |
| rx, ry, rw, rh, orig_w, orig_h = parts | |
| scale_x = width / float(orig_w) if orig_w > 0 else 1.0 | |
| scale_y = height / float(orig_h) if orig_h > 0 else 1.0 | |
| box_x = rx * scale_x | |
| box_y = ry * scale_y | |
| box_w = rw * scale_x | |
| box_h = rh * scale_y | |
| x = int(box_x + box_w / 2.0) | |
| y = int(box_y + box_h / 2.0) | |
| alignment_tag = r"\an5" | |
| align_code = 5 | |
| box_width_pct = max(35.0, min(98.0, (box_w / float(width)) * 100.0)) | |
| max_fitted_font = max(24, int(box_h * 0.70)) | |
| if font_size > max_fitted_font: | |
| font_size = max_fitted_font | |
| print(f"[RENDER] Auto-positioned Vietnamese subtitles directly inside blur region: pos=({x},{y}), box_w={box_w:.0f}, font_size={font_size}") | |
| else: | |
| x = int(width * float(caption_cfg.get("x_percent", 50)) / 100.0) | |
| y = int(height * float(caption_cfg.get("y_percent", 78)) / 100.0) | |
| box_width_pct = max(45.0, min(96.0, float(caption_cfg.get("box_width_percent", 86)))) | |
| except Exception as e: | |
| print(f"[RENDER] Warning parsing blur_region: {e}") | |
| x = int(width * float(caption_cfg.get("x_percent", 50)) / 100.0) | |
| y = int(height * float(caption_cfg.get("y_percent", 78)) / 100.0) | |
| box_width_pct = max(45.0, min(96.0, float(caption_cfg.get("box_width_percent", 86)))) | |
| else: | |
| x = int(width * float(caption_cfg.get("x_percent", 50)) / 100.0) | |
| y = int(height * float(caption_cfg.get("y_percent", 78)) / 100.0) | |
| box_width_pct = max(45.0, min(96.0, float(caption_cfg.get("box_width_percent", 86)))) | |
| # TikTok safe zone: ensure margins keep text inside 86% width / 78% height safe area | |
| # Bottom UI of TikTok covers ~ 12% height, right side buttons ~ 10% width β add safe padding | |
| safe_margin_v = max(38, int(height * 0.07)) # at least 7% from bottom/top | |
| safe_margin_h = max(28, int(width * 0.04)) | |
| margin = max(safe_margin_h, int((width * (100.0 - box_width_pct) / 100.0) / 2)) | |
| # Clamp y to stay within safe zone (avoid bottom 12% TikTok caption area) | |
| # y is calculated above; enforce ceiling at 82% of height for TikTok | |
| if 'y' in locals(): | |
| max_safe_y = int(height * 0.82) | |
| min_safe_y = int(height * 0.12) | |
| y = max(min_safe_y, min(max_safe_y, y)) | |
| else: | |
| # fallback safe y if not yet defined | |
| y = int(height * 0.78) | |
| uppercase = bool(caption_cfg.get("uppercase", False)) | |
| word_jump = bool(caption_cfg.get("word_jump", False)) | |
| colors = _theme_colors(caption_cfg.get("theme", "clean_white")) | |
| dialogue = [] | |
| for index, (start, end, text) in enumerate(events): | |
| cleaned_text = _compact_caption_text(text) | |
| if uppercase: | |
| cleaned_text = cleaned_text.upper() | |
| lines = _wrap_caption_words(cleaned_text, max_chars=max_chars, max_lines=max_lines) | |
| if not lines: | |
| continue | |
| cur_font_size = font_size | |
| total_len = len(cleaned_text) | |
| if len(lines) >= 3 or total_len > 45: | |
| cur_font_size = max(22, int(font_size * 0.85)) | |
| elif total_len > 30: | |
| cur_font_size = max(24, int(font_size * 0.92)) | |
| line_prefix = rf"{{{alignment_tag}\pos({x},{y})\fs{cur_font_size}\c{colors['primary']}\3c&H00000000&\bord{outline}\shad{shadow}" | |
| if word_jump: | |
| line_prefix += r"\fscx98\fscy98\t(0,95,\fscx110\fscy110)\t(95,230,\fscx100\fscy100)" | |
| line_prefix += "}" | |
| body = line_prefix + r"\N".join(_ass_escape(line) for line in lines) | |
| dialogue.append( | |
| f"Dialogue: 0,{_ass_time_from_srt(start)},{_ass_time_from_srt(end)},Caption,,{margin},{margin},0,,{body}" | |
| ) | |
| if not dialogue: | |
| return None | |
| doc = f"""[Script Info] | |
| ScriptType: v4.00+ | |
| PlayResX: {width} | |
| PlayResY: {height} | |
| ScaledBorderAndShadow: yes | |
| WrapStyle: 2 | |
| [V4+ Styles] | |
| Format: Name, Fontname, Fontsize, PrimaryColour, SecondaryColour, OutlineColour, BackColour, Bold, Italic, Underline, StrikeOut, ScaleX, ScaleY, Spacing, Angle, BorderStyle, Outline, Shadow, Alignment, MarginL, MarginR, MarginV, Encoding | |
| Style: Caption,{font_name},{font_size},{colors['primary']},{colors['secondary']},&H00000000,{shadow_color},-1,0,0,0,100,100,0.8,0,1,{outline},{shadow},{align_code},{margin},{margin},0,1 | |
| [Events] | |
| Format: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text | |
| {chr(10).join(dialogue)} | |
| """ | |
| ass_path = Path(ass_path) | |
| ass_path.parent.mkdir(parents=True, exist_ok=True) | |
| ass_path.write_text(doc, encoding="utf-8-sig") | |
| return ass_path | |
| def _region_to_native(blur_region, vid_w, vid_h): | |
| """Parse 'rx,ry,rw,rh,orig_w,orig_h' and re-derive the real blur/crop coords | |
| against the ACTUAL probed video dimensions using normalized fractions.""" | |
| parts = [int(p) for p in str(blur_region).split(',')] | |
| if len(parts) != 6: | |
| return None | |
| rx, ry, rw, rh, reg_orig_w, reg_orig_h = parts | |
| norm_x = rx / float(reg_orig_w) if reg_orig_w > 0 else 0.0 | |
| norm_y = ry / float(reg_orig_h) if reg_orig_h > 0 else 0.0 | |
| norm_w = rw / float(reg_orig_w) if reg_orig_w > 0 else 0.0 | |
| norm_h = rh / float(reg_orig_h) if reg_orig_h > 0 else 0.0 | |
| real_x = int(round(norm_x * vid_w)) | |
| real_y = int(round(norm_y * vid_h)) | |
| real_w = max(1, int(round(norm_w * vid_w))) | |
| real_h = max(1, int(round(norm_h * vid_h))) | |
| real_w = min(real_w, vid_w) | |
| real_h = min(real_h, vid_h) | |
| real_x = max(0, min(real_x, vid_w - real_w)) | |
| real_y = max(0, min(real_y, vid_h - real_h)) | |
| print(f"[RENDER] [Normalized Cords] x={norm_x:.4f}, y={norm_y:.4f}, w={norm_w:.4f}, h={norm_h:.4f}") | |
| print(f"[RENDER] [Native Frame Cords] x={real_x}, y={real_y}, w={real_w}, h={real_h} (frame {vid_w}x{vid_h})") | |
| return real_x, real_y, real_w, real_h, vid_w, vid_h | |
| def main(): | |
| parser = argparse.ArgumentParser(description="Standalone Video Render & Merge Worker CLI") | |
| parser.add_argument("--video", required=True, help="Path to input video file") | |
| parser.add_argument("--audio", required=True, help="Path to input mixed audio WAV file") | |
| parser.add_argument("--output", required=True, help="Path to output video file") | |
| parser.add_argument("--srt", default="", help="Path to translated SRT file for hardsubs") | |
| parser.add_argument("--blur-region", default="", help="Blur region as 'rx,ry,rw,rh,orig_w,orig_h'") | |
| parser.add_argument("--subtitle-region-mode", default="delogo", choices=["delogo", "blur", "none"], help="How to cover the selected source subtitle region") | |
| parser.add_argument("--preserve-regions", default="", help="Path to preserve_regions.json") | |
| parser.add_argument("--ffmpeg-path", default="ffmpeg", help="Path to ffmpeg executable") | |
| parser.add_argument("--encoder", default="auto", choices=["auto", "nvenc", "h264_nvenc", "libx264"], help="Video encoder to use") | |
| parser.add_argument("--allow-cpu-fallback", default="true", help="Allow fallback to CPU (true/false)") | |
| parser.add_argument("--zoom", default="1.0", help="Auto-zoom factor (e.g. 1.15 = zoom in 115%%), 1.0 disables") | |
| parser.add_argument("--trim-head-silence", default="0", help="Trim leading silence of video+audio by N seconds (0 disables)") | |
| parser.add_argument("--logo", default="", help="Path to logo image for overlay (PNG/JPG)") | |
| parser.add_argument("--logo-preset", default="bottom_right", help="Logo position preset") | |
| parser.add_argument("--logo-scale", default="15", help="Logo scale as %% of video width (5-40)") | |
| parser.add_argument("--logo-opacity", default="1.0", help="Logo opacity 0.0-1.0") | |
| parser.add_argument("--logo-x-pct", default="80", help="Logo custom X %%") | |
| parser.add_argument("--logo-y-pct", default="80", help="Logo custom Y %%") | |
| parser.add_argument("--logo-chromakey", default="true", help="Enable green chromakey removal (true/false)") | |
| parser.add_argument("--cta", default="", help="Path to CTA video for overlay (MP4/MOV with green bg)") | |
| parser.add_argument("--cta-preset", default="bottom_center", help="CTA position preset") | |
| parser.add_argument("--cta-scale", default="48", help="CTA scale as %% of video width (15-60) - TikTok-optimized default 48") | |
| parser.add_argument("--cta-opacity", default="1.0", help="CTA opacity 0.0-1.0") | |
| parser.add_argument("--cta-x-pct", default="50", help="CTA custom X %%") | |
| parser.add_argument("--cta-y-pct", default="85", help="CTA custom Y %%") | |
| parser.add_argument("--cta-interval", default="5.0", help="CTA appear interval seconds") | |
| parser.add_argument("--cta-duration", default="2.0", help="CTA visible duration per interval seconds") | |
| parser.add_argument("--cta-chromakey", default="true", help="Enable green chromakey for CTA (true/false)") | |
| args = parser.parse_args() | |
| video_path = Path(args.video) | |
| audio_path = Path(args.audio) | |
| output_path = Path(args.output) | |
| ffmpeg_path = Path(args.ffmpeg_path) | |
| if not video_path.exists(): | |
| print(f"Error: Input video not found at {video_path}", file=sys.stderr) | |
| sys.exit(1) | |
| if not audio_path.exists(): | |
| print(f"Error: Input audio not found at {audio_path}", file=sys.stderr) | |
| sys.exit(1) | |
| # 1. Parse preserve regions | |
| preserve_intervals = [] | |
| if args.preserve_regions: | |
| pr_path = Path(args.preserve_regions) | |
| if pr_path.exists(): | |
| try: | |
| with open(pr_path, "r", encoding="utf-8") as f: | |
| preserve_raw = json.load(f) | |
| # Convert ms to seconds | |
| preserve_intervals = [(x[0] / 1000.0, x[1] / 1000.0) for x in preserve_raw] | |
| print(f"Loaded preserve intervals: {len(preserve_intervals)} intervals.") | |
| except Exception as e: | |
| print(f"Warning: Failed to load preserve regions: {e}", file=sys.stderr) | |
| # Studio director: auto-zoom + trim leading silence | |
| zoom_factor = 1.0 | |
| try: | |
| zoom_factor = float(args.zoom) | |
| except Exception: | |
| zoom_factor = 1.0 | |
| trim_head_sec = 0.0 | |
| try: | |
| trim_head_sec = float(args.trim_head_silence) | |
| except Exception: | |
| trim_head_sec = 0.0 | |
| # When trimming head silence we must shift the SRT timeline BEFORE generating ASS. | |
| if trim_head_sec > 0 and args.srt: | |
| try: | |
| src_srt = Path(args.srt) | |
| shifted_srt = src_srt.with_name(src_srt.stem + "_shifted.srt") | |
| import shutil as _sh | |
| _sh.copy2(str(src_srt), str(shifted_srt)) | |
| _shift_srt(shifted_srt, trim_head_sec) | |
| args.srt = str(shifted_srt) | |
| print(f"[RENDER] Shifted subtitle timeline by -{trim_head_sec:.2f}s after head-silence trim.") | |
| except Exception as e: | |
| print(f"[RENDER] Warning: could not shift srt: {e}", file=sys.stderr) | |
| # ββ Logo & CTA detection ββ | |
| logo_enabled = bool(args.logo and Path(args.logo).exists()) | |
| logo_scale = 15.0 | |
| logo_opacity = 1.0 | |
| logo_preset = "bottom_right" | |
| logo_x_pct = 80.0 | |
| logo_y_pct = 80.0 | |
| logo_chromakey = True | |
| if logo_enabled: | |
| try: | |
| logo_scale = float(args.logo_scale) | |
| logo_opacity = float(args.logo_opacity) | |
| logo_preset = str(args.logo_preset) | |
| logo_x_pct = float(args.logo_x_pct) | |
| logo_y_pct = float(args.logo_y_pct) | |
| logo_chromakey = str(args.logo_chromakey).lower() in ("true","1","yes","t") | |
| print(f"[RENDER] Logo overlay: {Path(args.logo).name} preset={logo_preset} scale={logo_scale}% opacity={logo_opacity} chromakey={logo_chromakey}") | |
| except Exception as _e: | |
| print(f"[RENDER] Logo param parse error: {_e}", file=sys.stderr) | |
| cta_enabled = bool(args.cta and Path(args.cta).exists()) | |
| cta_scale = 35.0 | |
| cta_opacity = 1.0 | |
| cta_preset = "bottom_center" | |
| cta_x_pct = 50.0 | |
| cta_y_pct = 85.0 | |
| cta_interval = 5.0 | |
| cta_duration = 5.0 | |
| cta_chromakey = True | |
| if cta_enabled: | |
| try: | |
| cta_scale = float(args.cta_scale) | |
| cta_opacity = float(args.cta_opacity) | |
| cta_preset = str(args.cta_preset) | |
| cta_x_pct = float(args.cta_x_pct) | |
| cta_y_pct = float(args.cta_y_pct) | |
| cta_interval = float(args.cta_interval) | |
| cta_duration = float(args.cta_duration) | |
| cta_chromakey = str(args.cta_chromakey).lower() in ("true","1","yes","t") | |
| print(f"[RENDER] CTA overlay: {Path(args.cta).name} preset={cta_preset} scale={cta_scale}% every {cta_interval}s chromakey={cta_chromakey}") | |
| except Exception as _e: | |
| print(f"[RENDER] CTA param parse error: {_e}", file=sys.stderr) | |
| # 2. Build FFmpeg filter complex if blur or hardsub or logo or cta is needed | |
| filter_complex = None | |
| if args.blur_region or args.srt or zoom_factor > 1.0 or logo_enabled or cta_enabled: | |
| try: | |
| # Get video dimensions - try cv2 then ffprobe fallback for correct vertical/horizontal handling | |
| orig_w, orig_h = 1920, 1080 | |
| if args.video: | |
| try: | |
| import cv2 | |
| cap = cv2.VideoCapture(str(video_path)) | |
| w = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH)) or 0 | |
| h = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT)) or 0 | |
| cap.release() | |
| if w > 0 and h > 0: | |
| orig_w, orig_h = w, h | |
| else: | |
| raise ValueError("cv2 returned 0") | |
| except Exception: | |
| # ffprobe fallback (handles vertical videos correctly when cv2 unavailable) | |
| try: | |
| import subprocess as _sp, json as _js, shutil as _sh | |
| _ffprobe = _sh.which("ffprobe") or str(Path(args.ffmpeg_path).parent / "ffprobe.exe") if args.ffmpeg_path else "ffprobe" | |
| if not Path(_ffprobe).exists(): | |
| _ffprobe = "ffprobe" | |
| _res = _sp.run([_ffprobe, "-v", "error", "-select_streams", "v:0", "-show_entries", "stream=width,height", "-of", "json", str(video_path)], capture_output=True, text=True, timeout=5) | |
| _j = _js.loads(_res.stdout or "{}") | |
| _ws = _j.get("streams", [{}])[0] | |
| if _ws.get("width") and _ws.get("height"): | |
| orig_w, orig_h = int(_ws["width"]), int(_ws["height"]) | |
| except Exception: | |
| pass | |
| render_cfg = {} | |
| try: | |
| config_path = Path(__file__).parent.parent.parent / "config.json" | |
| if config_path.exists(): | |
| with open(config_path, "r", encoding="utf-8") as f: | |
| cfg_data = json.load(f) | |
| render_cfg = cfg_data.get("render", {}) | |
| except Exception as e: | |
| print(f"Warning: Failed to load render config: {e}", file=sys.stderr) | |
| dynamic_fontsize = 16 | |
| ass_playres_y = 288 | |
| ass_playres_x = int(288 * orig_w / orig_h) if orig_h else 384 | |
| if args.blur_region: | |
| converted = _region_to_native(args.blur_region, orig_w, orig_h) | |
| if converted is None: | |
| print("Error: invalid blur_region format, expected 'rx,ry,rw,rh,orig_w,orig_h'", file=sys.stderr) | |
| sys.exit(1) | |
| rx, ry, rw, rh, orig_w, orig_h = converted | |
| # TikTok-safe delogo inset: ensure 2px border inside frame to avoid "outside of frame" error | |
| # and guarantee even dimensions for yuv420p | |
| rx = max(2, rx) | |
| ry = max(2, ry) | |
| rw = min(rw, orig_w - rx - 2) | |
| rh = min(rh, orig_h - ry - 2) | |
| rx = rx if rx % 2 == 0 else max(0, rx - 1) | |
| ry = ry if ry % 2 == 0 else max(0, ry - 1) | |
| rw = rw if rw % 2 == 0 else rw - 1 | |
| rh = rh if rh % 2 == 0 else rh - 1 | |
| rw, rh = max(4, rw), max(4, rh) | |
| scale_x = ass_playres_x / orig_w if orig_w else 1 | |
| scale_y = ass_playres_y / orig_h if orig_h else 1 | |
| ass_rx = int(rx * scale_x) | |
| ass_ry = int(ry * scale_y) | |
| ass_rw = int(rw * scale_x) | |
| ass_rh = int(rh * scale_y) | |
| ass_margin_l = ass_rx | |
| ass_margin_r = max(0, ass_playres_x - (ass_rx + ass_rw)) | |
| disable_blur_on_preserve_regions = render_cfg.get("disable_blur_on_preserve_regions", True) | |
| if preserve_intervals and disable_blur_on_preserve_regions: | |
| ass_margin_v = 12 | |
| else: | |
| sub_block_h = int(2 * 1.35 * dynamic_fontsize) | |
| sub_block_h = min(sub_block_h, ass_rh) | |
| ass_bottom_of_text = ass_ry + ass_rh // 2 + sub_block_h // 2 | |
| ass_margin_v = max(0, ass_playres_y - ass_bottom_of_text) | |
| force_style = ( | |
| f"FontName=Arial,FontSize={dynamic_fontsize}," | |
| f"PrimaryColour=&H00FFFFFF,OutlineColour=&H00000000,BackColour=&H00000000," | |
| f"BorderStyle=1,Outline=2,Shadow=1,Alignment=2," | |
| f"MarginL={ass_margin_l},MarginR={ass_margin_r},MarginV={ass_margin_v}," | |
| f"WrapStyle=1" | |
| ) | |
| else: | |
| force_style = ( | |
| f"FontName=Arial,FontSize={dynamic_fontsize}," | |
| f"PrimaryColour=&H00FFFFFF,OutlineColour=&H00000000,BackColour=&H00000000," | |
| f"BorderStyle=1,Outline=2,Shadow=1,Alignment=2," | |
| f"MarginV=15,WrapStyle=1" | |
| ) | |
| # Build video blur / inpaint filter on original resolution | |
| video_filter = None | |
| if args.blur_region and args.subtitle_region_mode != "none": | |
| disable_blur_on_preserve_regions = render_cfg.get("disable_blur_on_preserve_regions", True) | |
| if preserve_intervals and disable_blur_on_preserve_regions: | |
| top_h = int(rh * 0.5) | |
| bot_y = ry + top_h | |
| bot_h = rh - top_h | |
| enable_str = "+".join( | |
| [f"between(t,{s:.3f},{e:.3f})" for s, e in preserve_intervals] | |
| ) | |
| if args.subtitle_region_mode == "blur": | |
| video_filter = ( | |
| f"[0:v]split[base][crop];" | |
| f"[crop]crop={rw}:{top_h}:{rx}:{ry},gblur=sigma=18[topblur];" | |
| f"[base][topblur]overlay={rx}:{ry}[v1];" | |
| f"[v1]split[base2][crop2];" | |
| f"[crop2]crop={rw}:{bot_h}:{rx}:{bot_y},gblur=sigma=18[botblur];" | |
| f"[base2][botblur]overlay={rx}:{bot_y}:enable='not({enable_str})'[blurred]" | |
| ) | |
| else: | |
| video_filter = ( | |
| f"[0:v]delogo=x={rx}:y={ry}:w={rw}:h={top_h}[v1];" | |
| f"[v1]delogo=x={rx}:y={bot_y}:w={rw}:h={bot_h}:enable='not({enable_str})'[blurred]" | |
| ) | |
| else: | |
| if args.subtitle_region_mode == "blur": | |
| video_filter = ( | |
| f"[0:v]split[base][crop];" | |
| f"[crop]crop={rw}:{rh}:{rx}:{ry},gblur=sigma=20[regionblur];" | |
| f"[base][regionblur]overlay={rx}:{ry}[blurred]" | |
| ) | |
| else: | |
| video_filter = f"[0:v]delogo=x={rx}:y={ry}:w={rw}:h={rh}[blurred]" | |
| # Build subtitle filter | |
| srt_filter = None | |
| if args.srt: | |
| caption_cfg = render_cfg.get("caption", {}) if isinstance(render_cfg, dict) else {} | |
| use_modern_ass = bool(caption_cfg.get("enabled", True)) | |
| ass_path = None | |
| if use_modern_ass: | |
| ass_blur_region = ( | |
| f"{rx},{ry},{rw},{rh},{orig_w},{orig_h}" if args.blur_region else None | |
| ) | |
| ass_path = _write_modern_ass_from_srt( | |
| args.srt, | |
| output_path.parent / "translated_caption.ass", | |
| orig_w, | |
| orig_h, | |
| caption_cfg, | |
| blur_region=ass_blur_region, | |
| ) | |
| if ass_path: | |
| ass_abs = str(Path(ass_path).resolve()).replace("\\", "/").replace(":", "\\:") | |
| srt_filter = f"subtitles='{ass_abs}'" | |
| print(f"[RENDER] modern ASS captions enabled: {ass_path}") | |
| else: | |
| srt_abs = str(Path(args.srt).resolve()).replace("\\", "/").replace(":", "\\:") | |
| srt_filter = f"subtitles='{srt_abs}':force_style='{force_style}'" | |
| # Auto-zoom filter string | |
| zoom_filter = None | |
| if zoom_factor > 1.0: | |
| zoom_filter = ( | |
| f"scale={int(orig_w*zoom_factor)}:{int(orig_h*zoom_factor)}:flags=lanczos," | |
| f"crop={orig_w}:{orig_h}:((iw-ow)/2):((ih-oh)/2),setsar=1" | |
| ) | |
| print(f"[RENDER] auto-zoom {zoom_factor}x enabled") | |
| # Combine filters: blur -> zoom -> logo -> subtitles | |
| # Build step-by-step with intermediate labels | |
| filter_parts = [] | |
| cur_label = None | |
| # Video blur/delogo (outputs [blurred] if exists) | |
| if video_filter: | |
| filter_parts.append(video_filter) | |
| cur_label = "blurred" | |
| else: | |
| cur_label = "0:v" | |
| # Zoom | |
| if zoom_filter: | |
| next_label = "v" | |
| # Reserve intermediate if logo/cta or subtitles will follow | |
| if logo_enabled or cta_enabled or srt_filter: | |
| next_label = "zm" | |
| filter_parts.append(f"[{cur_label}]{zoom_filter}[{next_label}]") | |
| cur_label = next_label | |
| # Logo overlay (logo is input 2) | |
| if logo_enabled: | |
| # Probe logo size for scale | |
| _lw, _lh = 2400, 1792 | |
| try: | |
| _img = cv2.imread(str(Path(args.logo)), cv2.IMREAD_UNCHANGED) | |
| if _img is not None: | |
| _lh, _lw = _img.shape[:2] | |
| except Exception: | |
| pass | |
| _target_w = orig_w * logo_scale / 100.0 | |
| _sf = max(0.02, min(0.5, _target_w / max(1, _lw))) | |
| # Clamp height to 85% of video height | |
| _target_h = _lh * _sf | |
| _max_h = orig_h * 0.85 | |
| if _target_h > _max_h: | |
| _sf = _max_h / max(1, _lh) | |
| _logo_vf_parts = [] | |
| if logo_chromakey: | |
| _logo_vf_parts.append("colorkey=0x00FF00:0.3:0.1") | |
| _logo_vf_parts.append("format=rgba") | |
| _logo_vf_parts.append(f"scale=iw*{_sf:.4f}:ih*{_sf:.4f}:flags=lanczos") | |
| if logo_opacity < 0.99: | |
| _logo_vf_parts.append(f"colorchannelmixer=aa={logo_opacity:.2f}") | |
| _logo_vf = ",".join(_logo_vf_parts) | |
| # Position | |
| _m = 2.0 | |
| if logo_preset == "top_left": | |
| _x, _y = f"W*{_m/100:.3f}", f"H*{_m/100:.3f}" | |
| elif logo_preset == "top_right": | |
| _x, _y = f"W-w-W*{_m/100:.3f}", f"H*{_m/100:.3f}" | |
| elif logo_preset == "bottom_left": | |
| _x, _y = f"W*{_m/100:.3f}", f"H-h-H*{_m/100:.3f}" | |
| elif logo_preset == "bottom_right": | |
| _x, _y = f"W-w-W*{_m/100:.3f}", f"H-h-H*{_m/100:.3f}" | |
| elif logo_preset == "center": | |
| _x, _y = "(W-w)/2", "(H-h)/2" | |
| elif logo_preset == "top_center": | |
| _x, _y = "(W-w)/2", f"H*{_m/100:.3f}" | |
| elif logo_preset == "bottom_center": | |
| _x, _y = "(W-w)/2", f"H-h-H*{_m/100:.3f}" | |
| elif logo_preset == "custom": | |
| _x, _y = f"W*{logo_x_pct/100:.4f}", f"H*{logo_y_pct/100:.4f}" | |
| else: | |
| _x, _y = f"W-w-W*{_m/100:.3f}", f"H-h-H*{_m/100:.3f}" | |
| filter_parts.append(f"[2:v]{_logo_vf}[logo]") | |
| next_label = "v" if not (cta_enabled or srt_filter) else "with_logo" | |
| filter_parts.append(f"[{cur_label}][logo]overlay={_x}:{_y}:format=rgb[{next_label}]") | |
| cur_label = next_label | |
| # CTA video overlay (periodic every interval, e.g., 5s) | |
| if cta_enabled: | |
| _cta_idx = 3 if logo_enabled else 2 | |
| # Probe CTA size | |
| _cw, _ch = 1080, 1920 | |
| try: | |
| _cap = cv2.VideoCapture(str(Path(args.cta))) | |
| _cw = int(_cap.get(cv2.CAP_PROP_FRAME_WIDTH)) or _cw | |
| _ch = int(_cap.get(cv2.CAP_PROP_FRAME_HEIGHT)) or _ch | |
| _cap.release() | |
| except Exception: | |
| pass | |
| # TikTok CTA bump: enforce minimum 42% width so CTA is clearly visible on mobile | |
| effective_cta_scale = max(float(cta_scale), 42.0) | |
| _cta_target_w = orig_w * effective_cta_scale / 100.0 | |
| _cta_sf = max(0.05, min(1.0, _cta_target_w / max(1, _cw))) | |
| _cta_target_h = _ch * _cta_sf | |
| _max_h2 = orig_h * 0.90 | |
| if _cta_target_h > _max_h2: | |
| _cta_sf = _max_h2 / max(1, _ch) | |
| _cta_vf_parts = [] | |
| if cta_chromakey: | |
| _cta_vf_parts.append("colorkey=0x00FF00:0.3:0.1") | |
| _cta_vf_parts.append("format=rgba") | |
| _cta_vf_parts.append(f"scale=iw*{_cta_sf:.4f}:ih*{_cta_sf:.4f}:flags=lanczos") | |
| if cta_opacity < 0.99: | |
| _cta_vf_parts.append(f"colorchannelmixer=aa={cta_opacity:.2f}") | |
| _cta_vf = ",".join(_cta_vf_parts) | |
| _m2 = 2.0 | |
| if cta_preset == "top_left": | |
| _cx, _cy = f"W*{_m2/100:.3f}", f"H*{_m2/100:.3f}" | |
| elif cta_preset == "top_right": | |
| _cx, _cy = f"W-w-W*{_m2/100:.3f}", f"H*{_m2/100:.3f}" | |
| elif cta_preset == "bottom_left": | |
| _cx, _cy = f"W*{_m2/100:.3f}", f"H-h-H*{_m2/100:.3f}" | |
| elif cta_preset == "bottom_right": | |
| _cx, _cy = f"W-w-W*{_m2/100:.3f}", f"H-h-H*{_m2/100:.3f}" | |
| elif cta_preset == "center": | |
| _cx, _cy = "(W-w)/2", "(H-h)/2" | |
| elif cta_preset == "top_center": | |
| _cx, _cy = "(W-w)/2", f"H*{_m2/100:.3f}" | |
| elif cta_preset == "bottom_center": | |
| _cx, _cy = "(W-w)/2", f"H-h-H*{_m2/100:.3f}" | |
| elif cta_preset == "custom": | |
| _cx, _cy = f"W*{cta_x_pct/100:.4f}", f"H*{cta_y_pct/100:.4f}" | |
| else: | |
| _cx, _cy = "(W-w)/2", f"H-h-H*{_m2/100:.3f}" | |
| _enable = f"lt(mod(t\\,{cta_interval})\\,{cta_duration})" | |
| filter_parts.append(f"[{_cta_idx}:v]{_cta_vf}[cta]") | |
| next_label2 = "v" if not srt_filter else "with_cta" | |
| filter_parts.append(f"[{cur_label}][cta]overlay={_cx}:{_cy}:format=rgb:enable='{_enable}'[{next_label2}]") | |
| cur_label = next_label2 | |
| # ββ TikTok polish: CFR 30fps + even dims + lanczos + light sharpen (before subs to keep text razor sharp) | |
| # This ensures sharp detail after blur/delogo/overlay without oversharpen | |
| tiktok_polish = "fps=30:round=near,scale=trunc(iw/2)*2:trunc(ih/2)*2:flags=lanczos+accurate_rnd+full_chroma_int:sws_dither=ed,unsharp=3:3:0.35:3:3:0.0" | |
| # Inject polish before subtitles (keep subtitles as final top layer) | |
| if srt_filter: | |
| # Need intermediate polish step | |
| if cur_label == "v": | |
| filter_parts.append(f"[{cur_label}]{tiktok_polish}[polished];[polished]{srt_filter}[v]") | |
| else: | |
| filter_parts.append(f"[{cur_label}]{tiktok_polish}[polished]") | |
| cur_label = "polished" | |
| filter_parts.append(f"[{cur_label}]{srt_filter}[v]") | |
| cur_label = "v" | |
| else: | |
| # No subtitles but still need TikTok CFR/sharpen/even polish for upload compliance | |
| if cur_label != "0:v": | |
| # Only add polish if we already have processing chain (avoid redundant filter on copy path) | |
| filter_parts.append(f"[{cur_label}]{tiktok_polish}[v]") | |
| cur_label = "v" | |
| # If filter_parts empty but logo only (no blur/zoom/srt) we already handled logo | |
| # If still no filter (should not happen), set to null | |
| if filter_parts: | |
| filter_complex = ";".join(filter_parts) | |
| print(f"[RENDER] filter complex pipeline: {filter_complex}") | |
| else: | |
| filter_complex = None | |
| except Exception as e: | |
| print(f"Error building filter complex: {e}", file=sys.stderr) | |
| sys.exit(1) | |
| allow_cpu_fallback = args.allow_cpu_fallback.lower() in ("true", "1", "yes", "t") | |
| # NVENC preflight check | |
| is_nvenc_requested = args.encoder in ("auto", "nvenc", "h264_nvenc") | |
| if is_nvenc_requested: | |
| print("Running NVENC preflight check...") | |
| if not check_nvenc_available(ffmpeg_path): | |
| print("NVENC preflight check failed.", file=sys.stderr) | |
| if not allow_cpu_fallback: | |
| print("NVENC unavailable or driver/API mismatch.\n" | |
| "CPU fallback disabled by strict GPU policy.\n" | |
| "Suggested fix: update NVIDIA driver or use FFmpeg build compatible with current driver.", file=sys.stderr) | |
| sys.exit(3) | |
| else: | |
| print("Warning: NVENC preflight check failed. Driver/API mismatch. CPU fallback enabled, falling back to libx264.") | |
| args.encoder = "libx264" | |
| # Resolve selected encoder for logging purposes | |
| selected_encoder = "h264_nvenc" if args.encoder in ("nvenc", "h264_nvenc") else (args.encoder if args.encoder != "auto" else "h264_nvenc") | |
| print(f"[RENDER] selected encoder: {selected_encoder}") | |
| print(f"[RENDER] cpu fallback: {'true' if allow_cpu_fallback else 'false'}") | |
| # 3. Determine encoders to try | |
| encoders_to_try = [] | |
| if args.encoder in ("nvenc", "h264_nvenc"): | |
| encoders_to_try = ["h264_nvenc"] | |
| if allow_cpu_fallback: | |
| encoders_to_try.append("libx264") | |
| elif args.encoder == "libx264": | |
| encoders_to_try = ["libx264"] | |
| else: # auto | |
| encoders_to_try = ["h264_nvenc"] | |
| if allow_cpu_fallback: | |
| encoders_to_try.append("libx264") | |
| output_path.parent.mkdir(parents=True, exist_ok=True) | |
| success = False | |
| for encoder in encoders_to_try: | |
| print(f"Attempting to render video using encoder '{encoder}'...") | |
| # TikTok spec: CFR, genpts, accurate seek, 48k AAC | |
| cmd = [ | |
| str(ffmpeg_path), "-y", "-noautorotate", | |
| "-fflags", "+genpts", | |
| "-avoid_negative_ts", "make_zero", | |
| ] | |
| if trim_head_sec > 0: | |
| cmd.extend(["-ss", f"{trim_head_sec:.3f}"]) | |
| cmd.extend(["-i", str(video_path)]) | |
| if trim_head_sec > 0: | |
| cmd.extend(["-ss", f"{trim_head_sec:.3f}"]) | |
| cmd.extend(["-i", str(audio_path)]) | |
| # Logo & CTA extra inputs (must be after video+audio so filter can ref [2:v]/[3:v]) | |
| _logo_enabled_cmd = bool(getattr(args, 'logo', '') and Path(args.logo).exists()) | |
| if _logo_enabled_cmd: | |
| cmd.extend(["-loop", "1", "-i", str(Path(args.logo))]) | |
| _cta_enabled_cmd = bool(getattr(args, 'cta', '') and Path(args.cta).exists()) | |
| if _cta_enabled_cmd: | |
| cmd.extend(["-stream_loop", "999", "-i", str(Path(args.cta))]) | |
| # ββ TikTok output compliance: ensure filter even if no blur/subs (CFR + sharpen) | |
| active_filter = filter_complex | |
| if not active_filter: | |
| # Minimal TikTok polish when no other processing: CFR 30 + even dims + light sharpen | |
| active_filter = "[0:v]fps=30:round=near,scale=trunc(iw/2)*2:trunc(ih/2)*2:flags=lanczos+accurate_rnd+full_chroma_int:sws_dither=ed,unsharp=3:3:0.35:3:3:0.0[v]" | |
| # Add -shortest when looped logo/CTA present to avoid infinite encode | |
| _need_shortest = logo_enabled or cta_enabled | |
| if _need_shortest: | |
| cmd.extend(["-filter_complex", active_filter, "-map", "[v]", "-map", "1:a:0", "-shortest"]) | |
| else: | |
| cmd.extend(["-filter_complex", active_filter, "-map", "[v]", "-map", "1:a:0"]) | |
| # ββ Video encoder: H.264 High Profile 8-bit yuv420p, CFR 30, high detail | |
| if encoder == "h264_nvenc": | |
| cmd.extend([ | |
| "-r", "30", | |
| "-c:v", "h264_nvenc", | |
| "-preset", "p4", "-tune", "hq", | |
| "-profile:v", "high", "-level", "4.1", | |
| "-rc", "vbr", "-cq", "19", "-b:v", "0", | |
| "-maxrate", "8M", "-bufsize", "12M", | |
| "-g", "60", "-keyint_min", "30", "-sc_threshold", "0", | |
| "-pix_fmt", "yuv420p", | |
| "-colorspace", "bt709", "-color_primaries", "bt709", "-color_trc", "bt709", "-color_range", "tv", | |
| ]) | |
| else: | |
| cmd.extend([ | |
| "-r", "30", | |
| "-c:v", "libx264", | |
| "-preset", "medium", "-crf", "18", | |
| "-profile:v", "high", "-level", "4.1", | |
| "-pix_fmt", "yuv420p", | |
| "-g", "60", "-keyint_min", "30", "-sc_threshold", "0", | |
| "-x264-params", "ref=4:bframes=2:me=hex:subme=7:psy=1:psy-rd=0.8:aq-mode=2", | |
| "-colorspace", "bt709", "-color_primaries", "bt709", "-color_trc", "bt709", "-color_range", "tv", | |
| ]) | |
| # ββ Audio: AAC-LC 48kHz 192k, 2ch, precise sync | |
| cmd.extend([ | |
| "-c:a", "aac", "-profile:a", "aac_low", | |
| "-ar", "48000", "-ac", "2", "-b:a", "192k", | |
| "-af", "aresample=async=1:min_hard_comp=0.100000:first_pts=0", | |
| ]) | |
| # ββ TikTok faststart + interleaving for AV sync & timestamp correctness | |
| cmd.extend([ | |
| "-movflags", "+faststart", | |
| "-fflags", "+genpts", | |
| "-max_interleave_delta", "100M", | |
| "-vsync", "cfr", "-fps_mode", "cfr", | |
| "-shortest", | |
| ]) | |
| cmd.append(str(output_path)) | |
| print(f"Executing: {' '.join(cmd)}") | |
| startupinfo = None | |
| if sys.platform == 'win32': | |
| startupinfo = subprocess.STARTUPINFO() | |
| startupinfo.dwFlags |= subprocess.STARTF_USESHOWWINDOW | |
| try: | |
| res = subprocess.run( | |
| cmd, | |
| stdout=subprocess.PIPE, | |
| stderr=subprocess.PIPE, | |
| text=True, | |
| encoding="utf-8", | |
| errors="ignore", | |
| startupinfo=startupinfo, | |
| timeout=1200 | |
| ) | |
| if res.returncode == 0: | |
| print(f"Render completed successfully using encoder '{encoder}'.") | |
| success = True | |
| break | |
| else: | |
| print(f"Warning: Encoder '{encoder}' failed with exit code {res.returncode}.", file=sys.stderr) | |
| print(f"FFmpeg Stderr:\n{res.stderr}", file=sys.stderr) | |
| except Exception as e: | |
| print(f"Warning: Exception using encoder '{encoder}': {e}", file=sys.stderr) | |
| if not success: | |
| if not allow_cpu_fallback and (args.encoder in ("nvenc", "h264_nvenc") or args.encoder == "auto"): | |
| print("NVENC requested but unavailable. CPU render fallback disabled by strict GPU policy.", file=sys.stderr) | |
| sys.exit(3) | |
| print("Error: All rendering encoders failed.", file=sys.stderr) | |
| sys.exit(2) | |
| sys.exit(0) | |
| if __name__ == "__main__": | |
| main() | |