""" app/core/studio_sfx.py ────────────────────── Retention-focused SFX injection for the studio pipeline. Generates short sound-effect beds (whoosh / ding / impact) with ffmpeg lavfi and mixes them onto the dubbed audio at the exact start_ms of every block that was tagged with an SFX marker by the LLM translation stage. Kept deliberately light (no external audio assets required) so it runs headless on any machine with ffmpeg. """ import json import subprocess import sys from pathlib import Path SFX_LENGTH_MS = { "whoosh": 500, "ding": 400, "impact": 300, "pop": 200, "boom": 600, } # volume in dB relative to the dubbed voice bed; SFX stays under the voice. SFX_GAIN_DB = { "whoosh": -16.0, "ding": -12.0, "impact": -10.0, "pop": -14.0, "boom": -9.0, } DEFAULT_SR = 44100 def _ffmpeg_cmd(ffmpeg_path, args): startupinfo = None if sys.platform == "win32": startupinfo = subprocess.STARTUPINFO() startupinfo.dwFlags |= subprocess.STARTF_USESHOWWINDOW cmd = [str(ffmpeg_path), "-y"] + args res = subprocess.run( cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, encoding="utf-8", errors="ignore", startupinfo=startupinfo, timeout=60, ) if res.returncode != 0: raise RuntimeError(f"ffmpeg failed: {res.stderr[-600:]}") def generate_sfx_clip(ffmpeg_path, sfx_name, output_path): """Render one sfx clip (mono 44.1k wav) using lavfi sources.""" sfx_name = str(sfx_name or "ding").lower() dur_ms = int(SFX_LENGTH_MS.get(sfx_name, 400)) dur = dur_ms / 1000.0 out = Path(output_path) if sfx_name == "whoosh": # filtered noise sweep upward, fade in/out -> airy whoosh src = ( f"anoisesrc=colour=white:amplitude=0.5:duration={dur}:sample_rate={DEFAULT_SR}," f"highpass=f=200,lowpass=f=4000," f"afade=t=in:st=0:d={dur*0.30:.3f},afade=t=out:st={dur*0.60:.3f}:d={dur*0.40:.3f}" ) elif sfx_name == "ding": # bright bell + short decay src = ( f"aevalsrc='0.35*sin(2*PI*1318*t)+0.15*sin(2*PI*2637*t)':d={dur}:s={DEFAULT_SR}," f"afade=t=in:st=0:d=0.01,afade=t=out:st={dur*0.25:.3f}:d={dur*0.75:.3f}" ) elif sfx_name == "boom": # low sub thump with punchy decay src = ( f"aevalsrc='0.6*sin(2*PI*55*t)*exp(-6*t)+0.2*sin(2*PI*110*t)*exp(-8*t)':d={dur}:s={DEFAULT_SR}," f"afade=t=in:st=0:d=0.01,afade=t=out:st={dur*0.30:.3f}:d={dur*0.70:.3f}" ) elif sfx_name == "pop": src = ( f"aevalsrc='0.4*sin(2*PI*900*t)*exp(-18*t)':d={dur}:s={DEFAULT_SR}," f"afade=t=out:st={dur*0.5:.3f}:d={dur*0.5:.3f}" ) else: # impact — mid-low thud src = ( f"aevalsrc='0.5*sin(2*PI*160*t)*exp(-9*t)+0.2*sin(2*PI*80*t)*exp(-6*t)':d={dur}:s={DEFAULT_SR}," f"afade=t=in:st=0:d=0.005,afade=t=out:st={dur*0.35:.3f}:d={dur*0.65:.3f}" ) out.parent.mkdir(parents=True, exist_ok=True) _ffmpeg_cmd(ffmpeg_path, ["-f", "lavfi", "-i", src, "-ac", "1", "-ar", str(DEFAULT_SR), str(out)]) return out def build_sfx_track(ffmpeg_path, tags_json, output_wav, total_duration_ms, sr=DEFAULT_SR): """Mix all tagged sfx into a single full-length track aligned to the video. tags_json: list of records from studio_tags.json (each has start_ms + sfx list). Returns the output path, or None when there are no sfx tags. """ tags_json = Path(tags_json) if not tags_json.exists(): return None with open(tags_json, "r", encoding="utf-8") as f: records = json.load(f) clips = [] for rec in records: sfx_list = [s for s in rec.get("sfx", []) if s in SFX_LENGTH_MS] if not sfx_list: continue start_ms = int(rec.get("start_ms", 0)) # stagger multiple sfx on the same line by 120ms each for i, sfx in enumerate(sfx_list): offset_ms = max(0, start_ms + i * 120) gain = SFX_GAIN_DB.get(sfx, -14.0) clips.append((sfx, offset_ms, gain)) if not clips: return None out = Path(output_wav) out.parent.mkdir(parents=True, exist_ok=True) temp_dir = out.parent / "sfx_tmp" temp_dir.mkdir(parents=True, exist_ok=True) mix_inputs = [] for i, (sfx, offset_ms, gain) in enumerate(clips): clip_wav = temp_dir / f"{i:03d}_{sfx}.wav" generate_sfx_clip(ffmpeg_path, sfx, clip_wav) delayed_wav = temp_dir / f"{i:03d}_delayed.wav" _ffmpeg_cmd( ffmpeg_path, [ "-i", str(clip_wav), "-af", f"adelay={offset_ms}|{offset_ms},volume={gain}dB", str(delayed_wav), ], ) mix_inputs.append(str(delayed_wav)) # amix all delayed clips into one track of the target length inputs = [] for w in mix_inputs: inputs += ["-i", w] n = len(mix_inputs) amix_inputs = "".join(f"[{i}:a]" for i in range(n)) _ffmpeg_cmd( ffmpeg_path, inputs + [ "-filter_complex", f"{amix_inputs}amix=inputs={n}:normalize=0", "-ar", str(sr), str(out), ], ) return out def mix_sfx_onto_bed(ffmpeg_path, bed_wav, sfx_track, output_wav): """Overlay the sfx track onto the dubbed audio bed (sfx stays subtle).""" bed = Path(bed_wav) track = Path(sfx_track) out = Path(output_wav) if not bed.exists() or not track.exists(): return str(bed) if bed.exists() else None out.parent.mkdir(parents=True, exist_ok=True) _ffmpeg_cmd( ffmpeg_path, [ "-i", str(bed), "-i", str(track), "-filter_complex", "[0:a][1:a]amix=inputs=2:normalize=0:dropout_transition=0", "-ar", str(DEFAULT_SR), str(out), ], ) return out