DRIPPY4 / app /core /studio_sfx.py
hoangtaiii's picture
Upload 67 files
21aadae verified
Raw History Blame Contribute Delete
6.05 kB
"""
app/core/studio_sfx.py
──────────────────────
Retention-focused SFX injection for the studio pipeline.
Generates short sound-effect beds (whoosh / ding / impact) with ffmpeg lavfi and
mixes them onto the dubbed audio at the exact start_ms of every block that was
tagged with an SFX marker by the LLM translation stage.
Kept deliberately light (no external audio assets required) so it runs headless
on any machine with ffmpeg.
"""
import json
import subprocess
import sys
from pathlib import Path
SFX_LENGTH_MS = {
"whoosh": 500,
"ding": 400,
"impact": 300,
"pop": 200,
"boom": 600,
}
# volume in dB relative to the dubbed voice bed; SFX stays under the voice.
SFX_GAIN_DB = {
"whoosh": -16.0,
"ding": -12.0,
"impact": -10.0,
"pop": -14.0,
"boom": -9.0,
}
DEFAULT_SR = 44100
def _ffmpeg_cmd(ffmpeg_path, args):
startupinfo = None
if sys.platform == "win32":
startupinfo = subprocess.STARTUPINFO()
startupinfo.dwFlags |= subprocess.STARTF_USESHOWWINDOW
cmd = [str(ffmpeg_path), "-y"] + args
res = subprocess.run(
cmd,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
text=True,
encoding="utf-8",
errors="ignore",
startupinfo=startupinfo,
timeout=60,
)
if res.returncode != 0:
raise RuntimeError(f"ffmpeg failed: {res.stderr[-600:]}")
def generate_sfx_clip(ffmpeg_path, sfx_name, output_path):
"""Render one sfx clip (mono 44.1k wav) using lavfi sources."""
sfx_name = str(sfx_name or "ding").lower()
dur_ms = int(SFX_LENGTH_MS.get(sfx_name, 400))
dur = dur_ms / 1000.0
out = Path(output_path)
if sfx_name == "whoosh":
# filtered noise sweep upward, fade in/out -> airy whoosh
src = (
f"anoisesrc=colour=white:amplitude=0.5:duration={dur}:sample_rate={DEFAULT_SR},"
f"highpass=f=200,lowpass=f=4000,"
f"afade=t=in:st=0:d={dur*0.30:.3f},afade=t=out:st={dur*0.60:.3f}:d={dur*0.40:.3f}"
)
elif sfx_name == "ding":
# bright bell + short decay
src = (
f"aevalsrc='0.35*sin(2*PI*1318*t)+0.15*sin(2*PI*2637*t)':d={dur}:s={DEFAULT_SR},"
f"afade=t=in:st=0:d=0.01,afade=t=out:st={dur*0.25:.3f}:d={dur*0.75:.3f}"
)
elif sfx_name == "boom":
# low sub thump with punchy decay
src = (
f"aevalsrc='0.6*sin(2*PI*55*t)*exp(-6*t)+0.2*sin(2*PI*110*t)*exp(-8*t)':d={dur}:s={DEFAULT_SR},"
f"afade=t=in:st=0:d=0.01,afade=t=out:st={dur*0.30:.3f}:d={dur*0.70:.3f}"
)
elif sfx_name == "pop":
src = (
f"aevalsrc='0.4*sin(2*PI*900*t)*exp(-18*t)':d={dur}:s={DEFAULT_SR},"
f"afade=t=out:st={dur*0.5:.3f}:d={dur*0.5:.3f}"
)
else: # impact — mid-low thud
src = (
f"aevalsrc='0.5*sin(2*PI*160*t)*exp(-9*t)+0.2*sin(2*PI*80*t)*exp(-6*t)':d={dur}:s={DEFAULT_SR},"
f"afade=t=in:st=0:d=0.005,afade=t=out:st={dur*0.35:.3f}:d={dur*0.65:.3f}"
)
out.parent.mkdir(parents=True, exist_ok=True)
_ffmpeg_cmd(ffmpeg_path, ["-f", "lavfi", "-i", src, "-ac", "1", "-ar", str(DEFAULT_SR), str(out)])
return out
def build_sfx_track(ffmpeg_path, tags_json, output_wav, total_duration_ms, sr=DEFAULT_SR):
"""Mix all tagged sfx into a single full-length track aligned to the video.
tags_json: list of records from studio_tags.json (each has start_ms + sfx list).
Returns the output path, or None when there are no sfx tags.
"""
tags_json = Path(tags_json)
if not tags_json.exists():
return None
with open(tags_json, "r", encoding="utf-8") as f:
records = json.load(f)
clips = []
for rec in records:
sfx_list = [s for s in rec.get("sfx", []) if s in SFX_LENGTH_MS]
if not sfx_list:
continue
start_ms = int(rec.get("start_ms", 0))
# stagger multiple sfx on the same line by 120ms each
for i, sfx in enumerate(sfx_list):
offset_ms = max(0, start_ms + i * 120)
gain = SFX_GAIN_DB.get(sfx, -14.0)
clips.append((sfx, offset_ms, gain))
if not clips:
return None
out = Path(output_wav)
out.parent.mkdir(parents=True, exist_ok=True)
temp_dir = out.parent / "sfx_tmp"
temp_dir.mkdir(parents=True, exist_ok=True)
mix_inputs = []
for i, (sfx, offset_ms, gain) in enumerate(clips):
clip_wav = temp_dir / f"{i:03d}_{sfx}.wav"
generate_sfx_clip(ffmpeg_path, sfx, clip_wav)
delayed_wav = temp_dir / f"{i:03d}_delayed.wav"
_ffmpeg_cmd(
ffmpeg_path,
[
"-i", str(clip_wav),
"-af", f"adelay={offset_ms}|{offset_ms},volume={gain}dB",
str(delayed_wav),
],
)
mix_inputs.append(str(delayed_wav))
# amix all delayed clips into one track of the target length
inputs = []
for w in mix_inputs:
inputs += ["-i", w]
n = len(mix_inputs)
amix_inputs = "".join(f"[{i}:a]" for i in range(n))
_ffmpeg_cmd(
ffmpeg_path,
inputs + [
"-filter_complex",
f"{amix_inputs}amix=inputs={n}:normalize=0",
"-ar", str(sr),
str(out),
],
)
return out
def mix_sfx_onto_bed(ffmpeg_path, bed_wav, sfx_track, output_wav):
"""Overlay the sfx track onto the dubbed audio bed (sfx stays subtle)."""
bed = Path(bed_wav)
track = Path(sfx_track)
out = Path(output_wav)
if not bed.exists() or not track.exists():
return str(bed) if bed.exists() else None
out.parent.mkdir(parents=True, exist_ok=True)
_ffmpeg_cmd(
ffmpeg_path,
[
"-i", str(bed),
"-i", str(track),
"-filter_complex",
"[0:a][1:a]amix=inputs=2:normalize=0:dropout_transition=0",
"-ar", str(DEFAULT_SR),
str(out),
],
)
return out