DRIPPY4 / app /core /render_worker_cli.py
hoangtaiii's picture
Upload 92 files
16c3ac7 verified
Raw History Blame Contribute Delete
44.3 kB
import sys
import os
import re
import json
import argparse
import subprocess
from pathlib import Path
# Enforce UTF-8 for Windows console
if sys.platform == 'win32':
try:
if hasattr(sys.stdout, 'reconfigure'):
sys.stdout.reconfigure(encoding='utf-8')
if hasattr(sys.stderr, 'reconfigure'):
sys.stderr.reconfigure(encoding='utf-8')
except Exception:
pass
def check_nvenc_available(ffmpeg_path):
cmd = [
str(ffmpeg_path), "-y",
"-f", "lavfi", "-i", "color=c=black:s=256x256",
"-t", "1",
"-c:v", "h264_nvenc",
"-f", "null", "-"
]
try:
startupinfo = None
if sys.platform == 'win32':
startupinfo = subprocess.STARTUPINFO()
startupinfo.dwFlags |= subprocess.STARTF_USESHOWWINDOW
res = subprocess.run(
cmd,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
text=True,
encoding="utf-8",
errors="ignore",
startupinfo=startupinfo,
timeout=10
)
return res.returncode == 0
except Exception:
return False
def _ass_time_from_srt(value):
m = re.match(r"^(\d{2}):(\d{2}):(\d{2})[,.](\d{3})$", str(value).strip())
if not m:
return "0:00:00.00"
hh, mm, ss, ms = [int(x) for x in m.groups()]
centis = int(round(ms / 10.0))
return f"{hh}:{mm:02d}:{ss:02d}.{centis:02d}"
def _parse_srt_events(srt_path):
try:
content = Path(srt_path).read_text(encoding="utf-8", errors="ignore").strip().replace("\r\n", "\n")
except Exception:
return []
pattern = r"(\d+)\s+(\d{2}:\d{2}:\d{2}[.,]\d{3})\s*-->\s*(\d{2}:\d{2}:\d{2}[.,]\d{3})\s*\n(.*?)(?=\n\s*\d+\s+\d{2}:\d{2}:\d{2}[.,]\d{3}\s*-->|\Z)"
events = []
for match in re.finditer(pattern, content, re.DOTALL):
text = " ".join(line.strip() for line in match.group(4).splitlines() if line.strip())
if text:
events.append((match.group(2), match.group(3), text))
return events
def _ass_escape(text):
return str(text or "").replace("\\", "\\\\").replace("{", r"\{").replace("}", r"\}").replace("\n", r"\N")
def _shift_srt(srt_path, seconds):
"""Shift every timestamp in an SRT file by -seconds (in place). Used after
trimming leading silence so hardsub captions stay aligned with the video."""
srt_path = Path(srt_path)
if seconds <= 0 or not srt_path.exists():
return
content = srt_path.read_text(encoding="utf-8", errors="ignore").replace("\r\n", "\n")
def shift_ts(m):
def to_ms(t):
tm = re.match(r"(\d+):(\d+):(\d+)[.,](\d+)", t)
if not tm:
return 0
hh, mm, ss, ms = (int(x) for x in tm.groups())
return ((hh * 3600 + mm * 60 + ss) * 1000) + ms
def from_ms(v):
v = max(0, int(v))
hh, rem = divmod(v, 3600000)
mm, rem = divmod(rem, 60000)
ss, ms = divmod(rem, 1000)
return f"{hh:02d}:{mm:02d}:{ss:02d},{ms:03d}"
delta = int(round(seconds * 1000))
return f"{from_ms(to_ms(m.group(1)) - delta)} --> {from_ms(to_ms(m.group(2)) - delta)}"
pattern = re.compile(r"(\d{2}:\d{2}:\d{2}[.,]\d{3})\s*-->\s*(\d{2}:\d{2}:\d{2}[.,]\d{3})")
new_content = pattern.sub(shift_ts, content)
srt_path.write_text(new_content, encoding="utf-8")
def _wrap_caption_words(text, max_chars=22, max_lines=3):
"""Wrap caption text naturally across lines without dropping words."""
words = str(text or "").split()
if not words:
return []
lines = []
current = ""
for word in words:
candidate = word if not current else f"{current} {word}"
if len(candidate) <= max_chars or not current:
current = candidate
else:
if len(lines) < max_lines - 1:
lines.append(current)
current = word
else:
# Ở dΓ²ng cuα»‘i cΓΉng, tiαΊΏp tα»₯c gα»™p toΓ n bα»™ tα»« cΓ²n lαΊ‘i vΓ o dΓ²ng nΓ y, khΓ΄ng được bỏ sΓ³t
current = f"{current} {word}"
if current:
lines.append(current)
# Xα»­ lΓ½ cΓ‘c tα»« nα»‘i khΓ΄ng Δ‘α»©ng lΖ‘ lα»­ng mα»™t mΓ¬nh ở cuα»‘i dΓ²ng
connectors = {"vΓ ", "vα»›i", "cα»§a", "lΓ ", "thΓ¬", "mΓ ", "rαΊ±ng", "nΓͺn", "nhΖ°ng", "hoαΊ·c", "vΓ¬"}
if len(lines) > 1:
for idx in range(len(lines) - 1):
parts = lines[idx].split()
if len(parts) > 1 and parts[-1].strip(" ,.!?;:").lower() in connectors:
moved_word = parts.pop()
lines[idx] = " ".join(parts)
lines[idx + 1] = f"{moved_word} {lines[idx + 1]}"
return lines[:max_lines]
def _compact_caption_text(text, max_words=None):
"""ChuαΊ©n hΓ³a phα»₯ đề tiαΊΏng Việt, giα»― nguyΓͺn vαΊΉn 100% nα»™i dung cΓ’u."""
text = str(text or "").strip()
text = re.sub(r"\s+", " ", text)
text = text.strip("\"'β€œβ€β€žβ€")
return text
def _theme_colors(theme):
theme = str(theme or "motion_drip").lower()
if theme == "clean_white":
return {"primary": "&H00FFFFFF", "secondary": "&H00FFFFFF"}
if theme == "brand_drip":
return {"primary": "&H00F2F2F2", "secondary": "&H00D7F7FF"}
if theme == "street_luxe":
return {"primary": "&H00E8F5FF", "secondary": "&H0099D6FF"}
return {"primary": "&H00F7F0DC", "secondary": "&H00A8F4FF"}
def _write_modern_ass_from_srt(srt_path, ass_path, width, height, caption_cfg, blur_region=None):
events = _parse_srt_events(srt_path)
if not events:
return None
# TikTok-optimized typography: larger, bolder, high contrast for mobile
# Vertical video needs bigger font; horizontal keeps moderate size
font_name = str(caption_cfg.get("font_name", "Arial Black, Montserrat, Be Vietnam Pro, Arial"))
# Increase default for TikTok readability: 82 vertical / 54 horizontal (was 68/48)
base_font_size = int(caption_cfg.get("font_size", 82 if height > width else 54))
max_chars = int(caption_cfg.get("max_chars_per_line", 22))
max_lines = int(caption_cfg.get("max_lines", 3))
# TikTok compression demands stronger outline/shadow for legibility
outline = float(caption_cfg.get("outline", 4.5))
shadow = float(caption_cfg.get("shadow", 2.2))
shadow_color = "&H99000000"
alignment_tag = r"\an8"
align_code = 8
font_size = base_font_size
if blur_region:
try:
parts = [int(p) for p in str(blur_region).split(',')]
if len(parts) == 6:
rx, ry, rw, rh, orig_w, orig_h = parts
scale_x = width / float(orig_w) if orig_w > 0 else 1.0
scale_y = height / float(orig_h) if orig_h > 0 else 1.0
box_x = rx * scale_x
box_y = ry * scale_y
box_w = rw * scale_x
box_h = rh * scale_y
x = int(box_x + box_w / 2.0)
y = int(box_y + box_h / 2.0)
alignment_tag = r"\an5"
align_code = 5
box_width_pct = max(35.0, min(98.0, (box_w / float(width)) * 100.0))
max_fitted_font = max(24, int(box_h * 0.70))
if font_size > max_fitted_font:
font_size = max_fitted_font
print(f"[RENDER] Auto-positioned Vietnamese subtitles directly inside blur region: pos=({x},{y}), box_w={box_w:.0f}, font_size={font_size}")
else:
x = int(width * float(caption_cfg.get("x_percent", 50)) / 100.0)
y = int(height * float(caption_cfg.get("y_percent", 78)) / 100.0)
box_width_pct = max(45.0, min(96.0, float(caption_cfg.get("box_width_percent", 86))))
except Exception as e:
print(f"[RENDER] Warning parsing blur_region: {e}")
x = int(width * float(caption_cfg.get("x_percent", 50)) / 100.0)
y = int(height * float(caption_cfg.get("y_percent", 78)) / 100.0)
box_width_pct = max(45.0, min(96.0, float(caption_cfg.get("box_width_percent", 86))))
else:
x = int(width * float(caption_cfg.get("x_percent", 50)) / 100.0)
y = int(height * float(caption_cfg.get("y_percent", 78)) / 100.0)
box_width_pct = max(45.0, min(96.0, float(caption_cfg.get("box_width_percent", 86))))
# TikTok safe zone: ensure margins keep text inside 86% width / 78% height safe area
# Bottom UI of TikTok covers ~ 12% height, right side buttons ~ 10% width β†’ add safe padding
safe_margin_v = max(38, int(height * 0.07)) # at least 7% from bottom/top
safe_margin_h = max(28, int(width * 0.04))
margin = max(safe_margin_h, int((width * (100.0 - box_width_pct) / 100.0) / 2))
# Clamp y to stay within safe zone (avoid bottom 12% TikTok caption area)
# y is calculated above; enforce ceiling at 82% of height for TikTok
if 'y' in locals():
max_safe_y = int(height * 0.82)
min_safe_y = int(height * 0.12)
y = max(min_safe_y, min(max_safe_y, y))
else:
# fallback safe y if not yet defined
y = int(height * 0.78)
uppercase = bool(caption_cfg.get("uppercase", False))
word_jump = bool(caption_cfg.get("word_jump", False))
colors = _theme_colors(caption_cfg.get("theme", "clean_white"))
dialogue = []
for index, (start, end, text) in enumerate(events):
cleaned_text = _compact_caption_text(text)
if uppercase:
cleaned_text = cleaned_text.upper()
lines = _wrap_caption_words(cleaned_text, max_chars=max_chars, max_lines=max_lines)
if not lines:
continue
cur_font_size = font_size
total_len = len(cleaned_text)
if len(lines) >= 3 or total_len > 45:
cur_font_size = max(22, int(font_size * 0.85))
elif total_len > 30:
cur_font_size = max(24, int(font_size * 0.92))
line_prefix = rf"{{{alignment_tag}\pos({x},{y})\fs{cur_font_size}\c{colors['primary']}\3c&H00000000&\bord{outline}\shad{shadow}"
if word_jump:
line_prefix += r"\fscx98\fscy98\t(0,95,\fscx110\fscy110)\t(95,230,\fscx100\fscy100)"
line_prefix += "}"
body = line_prefix + r"\N".join(_ass_escape(line) for line in lines)
dialogue.append(
f"Dialogue: 0,{_ass_time_from_srt(start)},{_ass_time_from_srt(end)},Caption,,{margin},{margin},0,,{body}"
)
if not dialogue:
return None
doc = f"""[Script Info]
ScriptType: v4.00+
PlayResX: {width}
PlayResY: {height}
ScaledBorderAndShadow: yes
WrapStyle: 2
[V4+ Styles]
Format: Name, Fontname, Fontsize, PrimaryColour, SecondaryColour, OutlineColour, BackColour, Bold, Italic, Underline, StrikeOut, ScaleX, ScaleY, Spacing, Angle, BorderStyle, Outline, Shadow, Alignment, MarginL, MarginR, MarginV, Encoding
Style: Caption,{font_name},{font_size},{colors['primary']},{colors['secondary']},&H00000000,{shadow_color},-1,0,0,0,100,100,0.8,0,1,{outline},{shadow},{align_code},{margin},{margin},0,1
[Events]
Format: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text
{chr(10).join(dialogue)}
"""
ass_path = Path(ass_path)
ass_path.parent.mkdir(parents=True, exist_ok=True)
ass_path.write_text(doc, encoding="utf-8-sig")
return ass_path
def _region_to_native(blur_region, vid_w, vid_h):
"""Parse 'rx,ry,rw,rh,orig_w,orig_h' and re-derive the real blur/crop coords
against the ACTUAL probed video dimensions using normalized fractions."""
parts = [int(p) for p in str(blur_region).split(',')]
if len(parts) != 6:
return None
rx, ry, rw, rh, reg_orig_w, reg_orig_h = parts
norm_x = rx / float(reg_orig_w) if reg_orig_w > 0 else 0.0
norm_y = ry / float(reg_orig_h) if reg_orig_h > 0 else 0.0
norm_w = rw / float(reg_orig_w) if reg_orig_w > 0 else 0.0
norm_h = rh / float(reg_orig_h) if reg_orig_h > 0 else 0.0
real_x = int(round(norm_x * vid_w))
real_y = int(round(norm_y * vid_h))
real_w = max(1, int(round(norm_w * vid_w)))
real_h = max(1, int(round(norm_h * vid_h)))
real_w = min(real_w, vid_w)
real_h = min(real_h, vid_h)
real_x = max(0, min(real_x, vid_w - real_w))
real_y = max(0, min(real_y, vid_h - real_h))
print(f"[RENDER] [Normalized Cords] x={norm_x:.4f}, y={norm_y:.4f}, w={norm_w:.4f}, h={norm_h:.4f}")
print(f"[RENDER] [Native Frame Cords] x={real_x}, y={real_y}, w={real_w}, h={real_h} (frame {vid_w}x{vid_h})")
return real_x, real_y, real_w, real_h, vid_w, vid_h
def main():
parser = argparse.ArgumentParser(description="Standalone Video Render & Merge Worker CLI")
parser.add_argument("--video", required=True, help="Path to input video file")
parser.add_argument("--audio", required=True, help="Path to input mixed audio WAV file")
parser.add_argument("--output", required=True, help="Path to output video file")
parser.add_argument("--srt", default="", help="Path to translated SRT file for hardsubs")
parser.add_argument("--blur-region", default="", help="Blur region as 'rx,ry,rw,rh,orig_w,orig_h'")
parser.add_argument("--subtitle-region-mode", default="delogo", choices=["delogo", "blur", "none"], help="How to cover the selected source subtitle region")
parser.add_argument("--preserve-regions", default="", help="Path to preserve_regions.json")
parser.add_argument("--ffmpeg-path", default="ffmpeg", help="Path to ffmpeg executable")
parser.add_argument("--encoder", default="auto", choices=["auto", "nvenc", "h264_nvenc", "libx264"], help="Video encoder to use")
parser.add_argument("--allow-cpu-fallback", default="true", help="Allow fallback to CPU (true/false)")
parser.add_argument("--zoom", default="1.0", help="Auto-zoom factor (e.g. 1.15 = zoom in 115%%), 1.0 disables")
parser.add_argument("--trim-head-silence", default="0", help="Trim leading silence of video+audio by N seconds (0 disables)")
parser.add_argument("--logo", default="", help="Path to logo image for overlay (PNG/JPG)")
parser.add_argument("--logo-preset", default="bottom_right", help="Logo position preset")
parser.add_argument("--logo-scale", default="15", help="Logo scale as %% of video width (5-40)")
parser.add_argument("--logo-opacity", default="1.0", help="Logo opacity 0.0-1.0")
parser.add_argument("--logo-x-pct", default="80", help="Logo custom X %%")
parser.add_argument("--logo-y-pct", default="80", help="Logo custom Y %%")
parser.add_argument("--logo-chromakey", default="true", help="Enable green chromakey removal (true/false)")
parser.add_argument("--cta", default="", help="Path to CTA video for overlay (MP4/MOV with green bg)")
parser.add_argument("--cta-preset", default="bottom_center", help="CTA position preset")
parser.add_argument("--cta-scale", default="48", help="CTA scale as %% of video width (15-60) - TikTok-optimized default 48")
parser.add_argument("--cta-opacity", default="1.0", help="CTA opacity 0.0-1.0")
parser.add_argument("--cta-x-pct", default="50", help="CTA custom X %%")
parser.add_argument("--cta-y-pct", default="85", help="CTA custom Y %%")
parser.add_argument("--cta-interval", default="5.0", help="CTA appear interval seconds")
parser.add_argument("--cta-duration", default="2.0", help="CTA visible duration per interval seconds")
parser.add_argument("--cta-chromakey", default="true", help="Enable green chromakey for CTA (true/false)")
args = parser.parse_args()
video_path = Path(args.video)
audio_path = Path(args.audio)
output_path = Path(args.output)
ffmpeg_path = Path(args.ffmpeg_path)
if not video_path.exists():
print(f"Error: Input video not found at {video_path}", file=sys.stderr)
sys.exit(1)
if not audio_path.exists():
print(f"Error: Input audio not found at {audio_path}", file=sys.stderr)
sys.exit(1)
# 1. Parse preserve regions
preserve_intervals = []
if args.preserve_regions:
pr_path = Path(args.preserve_regions)
if pr_path.exists():
try:
with open(pr_path, "r", encoding="utf-8") as f:
preserve_raw = json.load(f)
# Convert ms to seconds
preserve_intervals = [(x[0] / 1000.0, x[1] / 1000.0) for x in preserve_raw]
print(f"Loaded preserve intervals: {len(preserve_intervals)} intervals.")
except Exception as e:
print(f"Warning: Failed to load preserve regions: {e}", file=sys.stderr)
# Studio director: auto-zoom + trim leading silence
zoom_factor = 1.0
try:
zoom_factor = float(args.zoom)
except Exception:
zoom_factor = 1.0
trim_head_sec = 0.0
try:
trim_head_sec = float(args.trim_head_silence)
except Exception:
trim_head_sec = 0.0
# When trimming head silence we must shift the SRT timeline BEFORE generating ASS.
if trim_head_sec > 0 and args.srt:
try:
src_srt = Path(args.srt)
shifted_srt = src_srt.with_name(src_srt.stem + "_shifted.srt")
import shutil as _sh
_sh.copy2(str(src_srt), str(shifted_srt))
_shift_srt(shifted_srt, trim_head_sec)
args.srt = str(shifted_srt)
print(f"[RENDER] Shifted subtitle timeline by -{trim_head_sec:.2f}s after head-silence trim.")
except Exception as e:
print(f"[RENDER] Warning: could not shift srt: {e}", file=sys.stderr)
# ── Logo & CTA detection ──
logo_enabled = bool(args.logo and Path(args.logo).exists())
logo_scale = 15.0
logo_opacity = 1.0
logo_preset = "bottom_right"
logo_x_pct = 80.0
logo_y_pct = 80.0
logo_chromakey = True
if logo_enabled:
try:
logo_scale = float(args.logo_scale)
logo_opacity = float(args.logo_opacity)
logo_preset = str(args.logo_preset)
logo_x_pct = float(args.logo_x_pct)
logo_y_pct = float(args.logo_y_pct)
logo_chromakey = str(args.logo_chromakey).lower() in ("true","1","yes","t")
print(f"[RENDER] Logo overlay: {Path(args.logo).name} preset={logo_preset} scale={logo_scale}% opacity={logo_opacity} chromakey={logo_chromakey}")
except Exception as _e:
print(f"[RENDER] Logo param parse error: {_e}", file=sys.stderr)
cta_enabled = bool(args.cta and Path(args.cta).exists())
cta_scale = 35.0
cta_opacity = 1.0
cta_preset = "bottom_center"
cta_x_pct = 50.0
cta_y_pct = 85.0
cta_interval = 5.0
cta_duration = 5.0
cta_chromakey = True
if cta_enabled:
try:
cta_scale = float(args.cta_scale)
cta_opacity = float(args.cta_opacity)
cta_preset = str(args.cta_preset)
cta_x_pct = float(args.cta_x_pct)
cta_y_pct = float(args.cta_y_pct)
cta_interval = float(args.cta_interval)
cta_duration = float(args.cta_duration)
cta_chromakey = str(args.cta_chromakey).lower() in ("true","1","yes","t")
print(f"[RENDER] CTA overlay: {Path(args.cta).name} preset={cta_preset} scale={cta_scale}% every {cta_interval}s chromakey={cta_chromakey}")
except Exception as _e:
print(f"[RENDER] CTA param parse error: {_e}", file=sys.stderr)
# 2. Build FFmpeg filter complex if blur or hardsub or logo or cta is needed
filter_complex = None
if args.blur_region or args.srt or zoom_factor > 1.0 or logo_enabled or cta_enabled:
try:
# Get video dimensions - try cv2 then ffprobe fallback for correct vertical/horizontal handling
orig_w, orig_h = 1920, 1080
if args.video:
try:
import cv2
cap = cv2.VideoCapture(str(video_path))
w = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH)) or 0
h = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT)) or 0
cap.release()
if w > 0 and h > 0:
orig_w, orig_h = w, h
else:
raise ValueError("cv2 returned 0")
except Exception:
# ffprobe fallback (handles vertical videos correctly when cv2 unavailable)
try:
import subprocess as _sp, json as _js, shutil as _sh
_ffprobe = _sh.which("ffprobe") or str(Path(args.ffmpeg_path).parent / "ffprobe.exe") if args.ffmpeg_path else "ffprobe"
if not Path(_ffprobe).exists():
_ffprobe = "ffprobe"
_res = _sp.run([_ffprobe, "-v", "error", "-select_streams", "v:0", "-show_entries", "stream=width,height", "-of", "json", str(video_path)], capture_output=True, text=True, timeout=5)
_j = _js.loads(_res.stdout or "{}")
_ws = _j.get("streams", [{}])[0]
if _ws.get("width") and _ws.get("height"):
orig_w, orig_h = int(_ws["width"]), int(_ws["height"])
except Exception:
pass
render_cfg = {}
try:
config_path = Path(__file__).parent.parent.parent / "config.json"
if config_path.exists():
with open(config_path, "r", encoding="utf-8") as f:
cfg_data = json.load(f)
render_cfg = cfg_data.get("render", {})
except Exception as e:
print(f"Warning: Failed to load render config: {e}", file=sys.stderr)
dynamic_fontsize = 16
ass_playres_y = 288
ass_playres_x = int(288 * orig_w / orig_h) if orig_h else 384
if args.blur_region:
converted = _region_to_native(args.blur_region, orig_w, orig_h)
if converted is None:
print("Error: invalid blur_region format, expected 'rx,ry,rw,rh,orig_w,orig_h'", file=sys.stderr)
sys.exit(1)
rx, ry, rw, rh, orig_w, orig_h = converted
# TikTok-safe delogo inset: ensure 2px border inside frame to avoid "outside of frame" error
# and guarantee even dimensions for yuv420p
rx = max(2, rx)
ry = max(2, ry)
rw = min(rw, orig_w - rx - 2)
rh = min(rh, orig_h - ry - 2)
rx = rx if rx % 2 == 0 else max(0, rx - 1)
ry = ry if ry % 2 == 0 else max(0, ry - 1)
rw = rw if rw % 2 == 0 else rw - 1
rh = rh if rh % 2 == 0 else rh - 1
rw, rh = max(4, rw), max(4, rh)
scale_x = ass_playres_x / orig_w if orig_w else 1
scale_y = ass_playres_y / orig_h if orig_h else 1
ass_rx = int(rx * scale_x)
ass_ry = int(ry * scale_y)
ass_rw = int(rw * scale_x)
ass_rh = int(rh * scale_y)
ass_margin_l = ass_rx
ass_margin_r = max(0, ass_playres_x - (ass_rx + ass_rw))
disable_blur_on_preserve_regions = render_cfg.get("disable_blur_on_preserve_regions", True)
if preserve_intervals and disable_blur_on_preserve_regions:
ass_margin_v = 12
else:
sub_block_h = int(2 * 1.35 * dynamic_fontsize)
sub_block_h = min(sub_block_h, ass_rh)
ass_bottom_of_text = ass_ry + ass_rh // 2 + sub_block_h // 2
ass_margin_v = max(0, ass_playres_y - ass_bottom_of_text)
force_style = (
f"FontName=Arial,FontSize={dynamic_fontsize},"
f"PrimaryColour=&H00FFFFFF,OutlineColour=&H00000000,BackColour=&H00000000,"
f"BorderStyle=1,Outline=2,Shadow=1,Alignment=2,"
f"MarginL={ass_margin_l},MarginR={ass_margin_r},MarginV={ass_margin_v},"
f"WrapStyle=1"
)
else:
force_style = (
f"FontName=Arial,FontSize={dynamic_fontsize},"
f"PrimaryColour=&H00FFFFFF,OutlineColour=&H00000000,BackColour=&H00000000,"
f"BorderStyle=1,Outline=2,Shadow=1,Alignment=2,"
f"MarginV=15,WrapStyle=1"
)
# Build video blur / inpaint filter on original resolution
video_filter = None
if args.blur_region and args.subtitle_region_mode != "none":
disable_blur_on_preserve_regions = render_cfg.get("disable_blur_on_preserve_regions", True)
if preserve_intervals and disable_blur_on_preserve_regions:
top_h = int(rh * 0.5)
bot_y = ry + top_h
bot_h = rh - top_h
enable_str = "+".join(
[f"between(t,{s:.3f},{e:.3f})" for s, e in preserve_intervals]
)
if args.subtitle_region_mode == "blur":
video_filter = (
f"[0:v]split[base][crop];"
f"[crop]crop={rw}:{top_h}:{rx}:{ry},gblur=sigma=18[topblur];"
f"[base][topblur]overlay={rx}:{ry}[v1];"
f"[v1]split[base2][crop2];"
f"[crop2]crop={rw}:{bot_h}:{rx}:{bot_y},gblur=sigma=18[botblur];"
f"[base2][botblur]overlay={rx}:{bot_y}:enable='not({enable_str})'[blurred]"
)
else:
video_filter = (
f"[0:v]delogo=x={rx}:y={ry}:w={rw}:h={top_h}[v1];"
f"[v1]delogo=x={rx}:y={bot_y}:w={rw}:h={bot_h}:enable='not({enable_str})'[blurred]"
)
else:
if args.subtitle_region_mode == "blur":
video_filter = (
f"[0:v]split[base][crop];"
f"[crop]crop={rw}:{rh}:{rx}:{ry},gblur=sigma=20[regionblur];"
f"[base][regionblur]overlay={rx}:{ry}[blurred]"
)
else:
video_filter = f"[0:v]delogo=x={rx}:y={ry}:w={rw}:h={rh}[blurred]"
# Build subtitle filter
srt_filter = None
if args.srt:
caption_cfg = render_cfg.get("caption", {}) if isinstance(render_cfg, dict) else {}
use_modern_ass = bool(caption_cfg.get("enabled", True))
ass_path = None
if use_modern_ass:
ass_blur_region = (
f"{rx},{ry},{rw},{rh},{orig_w},{orig_h}" if args.blur_region else None
)
ass_path = _write_modern_ass_from_srt(
args.srt,
output_path.parent / "translated_caption.ass",
orig_w,
orig_h,
caption_cfg,
blur_region=ass_blur_region,
)
if ass_path:
ass_abs = str(Path(ass_path).resolve()).replace("\\", "/").replace(":", "\\:")
srt_filter = f"subtitles='{ass_abs}'"
print(f"[RENDER] modern ASS captions enabled: {ass_path}")
else:
srt_abs = str(Path(args.srt).resolve()).replace("\\", "/").replace(":", "\\:")
srt_filter = f"subtitles='{srt_abs}':force_style='{force_style}'"
# Auto-zoom filter string
zoom_filter = None
if zoom_factor > 1.0:
zoom_filter = (
f"scale={int(orig_w*zoom_factor)}:{int(orig_h*zoom_factor)}:flags=lanczos,"
f"crop={orig_w}:{orig_h}:((iw-ow)/2):((ih-oh)/2),setsar=1"
)
print(f"[RENDER] auto-zoom {zoom_factor}x enabled")
# Combine filters: blur -> zoom -> logo -> subtitles
# Build step-by-step with intermediate labels
filter_parts = []
cur_label = None
# Video blur/delogo (outputs [blurred] if exists)
if video_filter:
filter_parts.append(video_filter)
cur_label = "blurred"
else:
cur_label = "0:v"
# Zoom
if zoom_filter:
next_label = "v"
# Reserve intermediate if logo/cta or subtitles will follow
if logo_enabled or cta_enabled or srt_filter:
next_label = "zm"
filter_parts.append(f"[{cur_label}]{zoom_filter}[{next_label}]")
cur_label = next_label
# Logo overlay (logo is input 2)
if logo_enabled:
# Probe logo size for scale
_lw, _lh = 2400, 1792
try:
_img = cv2.imread(str(Path(args.logo)), cv2.IMREAD_UNCHANGED)
if _img is not None:
_lh, _lw = _img.shape[:2]
except Exception:
pass
_target_w = orig_w * logo_scale / 100.0
_sf = max(0.02, min(0.5, _target_w / max(1, _lw)))
# Clamp height to 85% of video height
_target_h = _lh * _sf
_max_h = orig_h * 0.85
if _target_h > _max_h:
_sf = _max_h / max(1, _lh)
_logo_vf_parts = []
if logo_chromakey:
_logo_vf_parts.append("colorkey=0x00FF00:0.3:0.1")
_logo_vf_parts.append("format=rgba")
_logo_vf_parts.append(f"scale=iw*{_sf:.4f}:ih*{_sf:.4f}:flags=lanczos")
if logo_opacity < 0.99:
_logo_vf_parts.append(f"colorchannelmixer=aa={logo_opacity:.2f}")
_logo_vf = ",".join(_logo_vf_parts)
# Position
_m = 2.0
if logo_preset == "top_left":
_x, _y = f"W*{_m/100:.3f}", f"H*{_m/100:.3f}"
elif logo_preset == "top_right":
_x, _y = f"W-w-W*{_m/100:.3f}", f"H*{_m/100:.3f}"
elif logo_preset == "bottom_left":
_x, _y = f"W*{_m/100:.3f}", f"H-h-H*{_m/100:.3f}"
elif logo_preset == "bottom_right":
_x, _y = f"W-w-W*{_m/100:.3f}", f"H-h-H*{_m/100:.3f}"
elif logo_preset == "center":
_x, _y = "(W-w)/2", "(H-h)/2"
elif logo_preset == "top_center":
_x, _y = "(W-w)/2", f"H*{_m/100:.3f}"
elif logo_preset == "bottom_center":
_x, _y = "(W-w)/2", f"H-h-H*{_m/100:.3f}"
elif logo_preset == "custom":
_x, _y = f"W*{logo_x_pct/100:.4f}", f"H*{logo_y_pct/100:.4f}"
else:
_x, _y = f"W-w-W*{_m/100:.3f}", f"H-h-H*{_m/100:.3f}"
filter_parts.append(f"[2:v]{_logo_vf}[logo]")
next_label = "v" if not (cta_enabled or srt_filter) else "with_logo"
filter_parts.append(f"[{cur_label}][logo]overlay={_x}:{_y}:format=rgb[{next_label}]")
cur_label = next_label
# CTA video overlay (periodic every interval, e.g., 5s)
if cta_enabled:
_cta_idx = 3 if logo_enabled else 2
# Probe CTA size
_cw, _ch = 1080, 1920
try:
_cap = cv2.VideoCapture(str(Path(args.cta)))
_cw = int(_cap.get(cv2.CAP_PROP_FRAME_WIDTH)) or _cw
_ch = int(_cap.get(cv2.CAP_PROP_FRAME_HEIGHT)) or _ch
_cap.release()
except Exception:
pass
# TikTok CTA bump: enforce minimum 42% width so CTA is clearly visible on mobile
effective_cta_scale = max(float(cta_scale), 42.0)
_cta_target_w = orig_w * effective_cta_scale / 100.0
_cta_sf = max(0.05, min(1.0, _cta_target_w / max(1, _cw)))
_cta_target_h = _ch * _cta_sf
_max_h2 = orig_h * 0.90
if _cta_target_h > _max_h2:
_cta_sf = _max_h2 / max(1, _ch)
_cta_vf_parts = []
if cta_chromakey:
_cta_vf_parts.append("colorkey=0x00FF00:0.3:0.1")
_cta_vf_parts.append("format=rgba")
_cta_vf_parts.append(f"scale=iw*{_cta_sf:.4f}:ih*{_cta_sf:.4f}:flags=lanczos")
if cta_opacity < 0.99:
_cta_vf_parts.append(f"colorchannelmixer=aa={cta_opacity:.2f}")
_cta_vf = ",".join(_cta_vf_parts)
_m2 = 2.0
if cta_preset == "top_left":
_cx, _cy = f"W*{_m2/100:.3f}", f"H*{_m2/100:.3f}"
elif cta_preset == "top_right":
_cx, _cy = f"W-w-W*{_m2/100:.3f}", f"H*{_m2/100:.3f}"
elif cta_preset == "bottom_left":
_cx, _cy = f"W*{_m2/100:.3f}", f"H-h-H*{_m2/100:.3f}"
elif cta_preset == "bottom_right":
_cx, _cy = f"W-w-W*{_m2/100:.3f}", f"H-h-H*{_m2/100:.3f}"
elif cta_preset == "center":
_cx, _cy = "(W-w)/2", "(H-h)/2"
elif cta_preset == "top_center":
_cx, _cy = "(W-w)/2", f"H*{_m2/100:.3f}"
elif cta_preset == "bottom_center":
_cx, _cy = "(W-w)/2", f"H-h-H*{_m2/100:.3f}"
elif cta_preset == "custom":
_cx, _cy = f"W*{cta_x_pct/100:.4f}", f"H*{cta_y_pct/100:.4f}"
else:
_cx, _cy = "(W-w)/2", f"H-h-H*{_m2/100:.3f}"
_enable = f"lt(mod(t\\,{cta_interval})\\,{cta_duration})"
filter_parts.append(f"[{_cta_idx}:v]{_cta_vf}[cta]")
next_label2 = "v" if not srt_filter else "with_cta"
filter_parts.append(f"[{cur_label}][cta]overlay={_cx}:{_cy}:format=rgb:enable='{_enable}'[{next_label2}]")
cur_label = next_label2
# ── TikTok polish: CFR 30fps + even dims + lanczos + light sharpen (before subs to keep text razor sharp)
# This ensures sharp detail after blur/delogo/overlay without oversharpen
tiktok_polish = "fps=30:round=near,scale=trunc(iw/2)*2:trunc(ih/2)*2:flags=lanczos+accurate_rnd+full_chroma_int:sws_dither=ed,unsharp=3:3:0.35:3:3:0.0"
# Inject polish before subtitles (keep subtitles as final top layer)
if srt_filter:
# Need intermediate polish step
if cur_label == "v":
filter_parts.append(f"[{cur_label}]{tiktok_polish}[polished];[polished]{srt_filter}[v]")
else:
filter_parts.append(f"[{cur_label}]{tiktok_polish}[polished]")
cur_label = "polished"
filter_parts.append(f"[{cur_label}]{srt_filter}[v]")
cur_label = "v"
else:
# No subtitles but still need TikTok CFR/sharpen/even polish for upload compliance
if cur_label != "0:v":
# Only add polish if we already have processing chain (avoid redundant filter on copy path)
filter_parts.append(f"[{cur_label}]{tiktok_polish}[v]")
cur_label = "v"
# If filter_parts empty but logo only (no blur/zoom/srt) we already handled logo
# If still no filter (should not happen), set to null
if filter_parts:
filter_complex = ";".join(filter_parts)
print(f"[RENDER] filter complex pipeline: {filter_complex}")
else:
filter_complex = None
except Exception as e:
print(f"Error building filter complex: {e}", file=sys.stderr)
sys.exit(1)
allow_cpu_fallback = args.allow_cpu_fallback.lower() in ("true", "1", "yes", "t")
# NVENC preflight check
is_nvenc_requested = args.encoder in ("auto", "nvenc", "h264_nvenc")
if is_nvenc_requested:
print("Running NVENC preflight check...")
if not check_nvenc_available(ffmpeg_path):
print("NVENC preflight check failed.", file=sys.stderr)
if not allow_cpu_fallback:
print("NVENC unavailable or driver/API mismatch.\n"
"CPU fallback disabled by strict GPU policy.\n"
"Suggested fix: update NVIDIA driver or use FFmpeg build compatible with current driver.", file=sys.stderr)
sys.exit(3)
else:
print("Warning: NVENC preflight check failed. Driver/API mismatch. CPU fallback enabled, falling back to libx264.")
args.encoder = "libx264"
# Resolve selected encoder for logging purposes
selected_encoder = "h264_nvenc" if args.encoder in ("nvenc", "h264_nvenc") else (args.encoder if args.encoder != "auto" else "h264_nvenc")
print(f"[RENDER] selected encoder: {selected_encoder}")
print(f"[RENDER] cpu fallback: {'true' if allow_cpu_fallback else 'false'}")
# 3. Determine encoders to try
encoders_to_try = []
if args.encoder in ("nvenc", "h264_nvenc"):
encoders_to_try = ["h264_nvenc"]
if allow_cpu_fallback:
encoders_to_try.append("libx264")
elif args.encoder == "libx264":
encoders_to_try = ["libx264"]
else: # auto
encoders_to_try = ["h264_nvenc"]
if allow_cpu_fallback:
encoders_to_try.append("libx264")
output_path.parent.mkdir(parents=True, exist_ok=True)
success = False
for encoder in encoders_to_try:
print(f"Attempting to render video using encoder '{encoder}'...")
# TikTok spec: CFR, genpts, accurate seek, 48k AAC
cmd = [
str(ffmpeg_path), "-y", "-noautorotate",
"-fflags", "+genpts",
"-avoid_negative_ts", "make_zero",
]
if trim_head_sec > 0:
cmd.extend(["-ss", f"{trim_head_sec:.3f}"])
cmd.extend(["-i", str(video_path)])
if trim_head_sec > 0:
cmd.extend(["-ss", f"{trim_head_sec:.3f}"])
cmd.extend(["-i", str(audio_path)])
# Logo & CTA extra inputs (must be after video+audio so filter can ref [2:v]/[3:v])
_logo_enabled_cmd = bool(getattr(args, 'logo', '') and Path(args.logo).exists())
if _logo_enabled_cmd:
cmd.extend(["-loop", "1", "-i", str(Path(args.logo))])
_cta_enabled_cmd = bool(getattr(args, 'cta', '') and Path(args.cta).exists())
if _cta_enabled_cmd:
cmd.extend(["-stream_loop", "999", "-i", str(Path(args.cta))])
# ── TikTok output compliance: ensure filter even if no blur/subs (CFR + sharpen)
active_filter = filter_complex
if not active_filter:
# Minimal TikTok polish when no other processing: CFR 30 + even dims + light sharpen
active_filter = "[0:v]fps=30:round=near,scale=trunc(iw/2)*2:trunc(ih/2)*2:flags=lanczos+accurate_rnd+full_chroma_int:sws_dither=ed,unsharp=3:3:0.35:3:3:0.0[v]"
# Add -shortest when looped logo/CTA present to avoid infinite encode
_need_shortest = logo_enabled or cta_enabled
if _need_shortest:
cmd.extend(["-filter_complex", active_filter, "-map", "[v]", "-map", "1:a:0", "-shortest"])
else:
cmd.extend(["-filter_complex", active_filter, "-map", "[v]", "-map", "1:a:0"])
# ── Video encoder: H.264 High Profile 8-bit yuv420p, CFR 30, high detail
if encoder == "h264_nvenc":
cmd.extend([
"-r", "30",
"-c:v", "h264_nvenc",
"-preset", "p4", "-tune", "hq",
"-profile:v", "high", "-level", "4.1",
"-rc", "vbr", "-cq", "19", "-b:v", "0",
"-maxrate", "8M", "-bufsize", "12M",
"-g", "60", "-keyint_min", "30", "-sc_threshold", "0",
"-pix_fmt", "yuv420p",
"-colorspace", "bt709", "-color_primaries", "bt709", "-color_trc", "bt709", "-color_range", "tv",
])
else:
cmd.extend([
"-r", "30",
"-c:v", "libx264",
"-preset", "medium", "-crf", "18",
"-profile:v", "high", "-level", "4.1",
"-pix_fmt", "yuv420p",
"-g", "60", "-keyint_min", "30", "-sc_threshold", "0",
"-x264-params", "ref=4:bframes=2:me=hex:subme=7:psy=1:psy-rd=0.8:aq-mode=2",
"-colorspace", "bt709", "-color_primaries", "bt709", "-color_trc", "bt709", "-color_range", "tv",
])
# ── Audio: AAC-LC 48kHz 192k, 2ch, precise sync
cmd.extend([
"-c:a", "aac", "-profile:a", "aac_low",
"-ar", "48000", "-ac", "2", "-b:a", "192k",
"-af", "aresample=async=1:min_hard_comp=0.100000:first_pts=0",
])
# ── TikTok faststart + interleaving for AV sync & timestamp correctness
cmd.extend([
"-movflags", "+faststart",
"-fflags", "+genpts",
"-max_interleave_delta", "100M",
"-vsync", "cfr", "-fps_mode", "cfr",
"-shortest",
])
cmd.append(str(output_path))
print(f"Executing: {' '.join(cmd)}")
startupinfo = None
if sys.platform == 'win32':
startupinfo = subprocess.STARTUPINFO()
startupinfo.dwFlags |= subprocess.STARTF_USESHOWWINDOW
try:
res = subprocess.run(
cmd,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
text=True,
encoding="utf-8",
errors="ignore",
startupinfo=startupinfo,
timeout=1200
)
if res.returncode == 0:
print(f"Render completed successfully using encoder '{encoder}'.")
success = True
break
else:
print(f"Warning: Encoder '{encoder}' failed with exit code {res.returncode}.", file=sys.stderr)
print(f"FFmpeg Stderr:\n{res.stderr}", file=sys.stderr)
except Exception as e:
print(f"Warning: Exception using encoder '{encoder}': {e}", file=sys.stderr)
if not success:
if not allow_cpu_fallback and (args.encoder in ("nvenc", "h264_nvenc") or args.encoder == "auto"):
print("NVENC requested but unavailable. CPU render fallback disabled by strict GPU policy.", file=sys.stderr)
sys.exit(3)
print("Error: All rendering encoders failed.", file=sys.stderr)
sys.exit(2)
sys.exit(0)
if __name__ == "__main__":
main()