Spaces:
Running on Zero
Running on Zero
Download app/core/audio_mixer.py from hoangtaiii/DRIPPY4: direct link, hf CLI and curl.
- Browser
- Download file 11.7 kB
-
https://huggingface.co/spaces/hoangtaiii/DRIPPY4/resolve/main/app/core/audio_mixer.py
- Command line
-
hf download hf://spaces/hoangtaiii/DRIPPY4/app/core/audio_mixer.py
-
curl -L -o audio_mixer.py https://huggingface.co/spaces/hoangtaiii/DRIPPY4/resolve/main/app/core/audio_mixer.py
11.7 kB
| import sys | |
| import math | |
| from pydub import AudioSegment | |
| class AudioMixer: | |
| def __init__(self, ffmpeg_path="ffmpeg"): | |
| AudioSegment.converter = str(ffmpeg_path) | |
| def mix_dubbed_audio(self, background_path, vocals_path, original_audio_path, tts_segments, preserve_intervals, output_path, allow_degraded_fallback=True, segment_actions=None, chinese_vocal_handling="lower", chinese_vocal_volume_percent=35, vi_voiceover_enabled=True, vi_voiceover_volume_percent=100, mute_original_audio=False): | |
| """ | |
| Mixes background track, TTS audio segments, and original vocals for preserved zones. | |
| Applies soft ducking and 80ms crossfades for transitions. | |
| """ | |
| print(f"Mixing stems: bg={background_path}, vocals={vocals_path}, orig={original_audio_path}") | |
| # Load audio stems | |
| import os | |
| from pathlib import Path | |
| bg_path = Path(background_path) if background_path else None | |
| if bg_path and bg_path.exists(): | |
| bg_audio = AudioSegment.from_file(str(bg_path)) | |
| else: | |
| if not allow_degraded_fallback: | |
| raise Exception("Background stem missing and degraded original audio fallback disabled.") | |
| print("Warning: Background audio stem not found, falling back to original audio.") | |
| bg_audio = AudioSegment.from_file(str(original_audio_path)) | |
| if vocals_path and vocals_path.exists(): | |
| vocals_audio = AudioSegment.from_file(vocals_path) | |
| else: | |
| vocals_audio = AudioSegment.from_file(original_audio_path) | |
| original_audio = AudioSegment.from_file(str(original_audio_path)) if original_audio_path and Path(original_audio_path).exists() else None | |
| if original_audio is None: | |
| if not allow_degraded_fallback: | |
| raise Exception("Original audio missing and degraded fallback disabled.") | |
| original_audio = bg_audio | |
| total_duration_ms = len(original_audio) | |
| bg_audio = self._fit_duration(bg_audio, total_duration_ms) | |
| vocals_audio = self._fit_duration(vocals_audio, total_duration_ms) | |
| original_audio = self._fit_duration(original_audio, total_duration_ms) | |
| # ── MUTE ORIGINAL: chỉ giữ giọng Việt, tắt toàn bộ audio gốc ── | |
| mute_original_audio = bool(mute_original_audio) | |
| # Voice overlay stream | |
| voice_overlays = AudioSegment.silent(duration=total_duration_ms) | |
| voice_active_mask = [False] * total_duration_ms | |
| vi_voiceover_enabled = bool(vi_voiceover_enabled) and float(vi_voiceover_volume_percent) > 0 | |
| vi_gain_db = self._gain_db_from_percent(vi_voiceover_volume_percent) | |
| # Overlay each TTS segment | |
| for block_id, seg_path in tts_segments.items(): | |
| if not vi_voiceover_enabled: | |
| continue | |
| if not seg_path.exists(): | |
| continue | |
| segment = AudioSegment.from_file(seg_path) | |
| if vi_gain_db is not None: | |
| segment = segment + vi_gain_db | |
| # Find start_ms from filename or block info | |
| # The tts segments should be passed as {start_ms: seg_path} | |
| try: | |
| start_ms = int(block_id) | |
| except ValueError: | |
| continue | |
| if start_ms >= total_duration_ms: | |
| continue | |
| # Overlay segment | |
| voice_overlays = voice_overlays.overlay(segment, position=start_ms) | |
| # Mark mask | |
| seg_len = len(segment) | |
| for m in range(start_ms, min(total_duration_ms, start_ms + seg_len)): | |
| voice_active_mask[m] = True | |
| if mute_original_audio: | |
| # Tắt hoàn toàn tiếng gốc: base là silence, chỉ overlay TTS Việt | |
| print(f"[MUTE ORIGINAL] Tắt toàn bộ âm thanh gốc, chỉ giữ giọng Việt ({len(tts_segments)} segments)") | |
| for block_id, seg_path in tts_segments.items(): | |
| if not vi_voiceover_enabled: | |
| break | |
| if not seg_path.exists(): | |
| continue | |
| try: | |
| start_ms = int(block_id) | |
| except ValueError: | |
| continue | |
| if start_ms >= total_duration_ms: | |
| continue | |
| # voice_overlays đã được build ở trên, ở chế độ mute ta tạo final từ silence + voice | |
| final_audio = AudioSegment.silent(duration=total_duration_ms) | |
| if vi_voiceover_enabled: | |
| final_audio = final_audio.overlay(voice_overlays) | |
| # Nếu không có TTS nào (ví dụ NEED_REVIEW), vẫn export silence để giữ duration | |
| output_path.parent.mkdir(parents=True, exist_ok=True) | |
| final_audio.export(str(output_path), format="wav") | |
| print(f"[MUTE ORIGINAL] Mixed track (silence + VI voice) exported to {output_path}") | |
| return True | |
| if segment_actions: | |
| final_audio = self._mix_language_aware( | |
| original_audio=original_audio, | |
| background_audio=bg_audio, | |
| vocals_audio=vocals_audio, | |
| voice_overlays=voice_overlays, | |
| segment_actions=segment_actions, | |
| chinese_vocal_handling=chinese_vocal_handling, | |
| chinese_vocal_volume_percent=chinese_vocal_volume_percent, | |
| total_duration_ms=total_duration_ms, | |
| ) | |
| output_path.parent.mkdir(parents=True, exist_ok=True) | |
| final_audio.export(str(output_path), format="wav") | |
| print(f"Language-aware mixed track exported successfully to {output_path}") | |
| return True | |
| # Legacy fallback: Group contiguous active voice intervals for ducking | |
| active_intervals = [] | |
| in_active = False | |
| start_act = 0 | |
| for m in range(total_duration_ms): | |
| if voice_active_mask[m] and not in_active: | |
| in_active = True | |
| start_act = m | |
| elif not voice_active_mask[m] and in_active: | |
| in_active = False | |
| active_intervals.append((start_act, m)) | |
| if in_active: | |
| active_intervals.append((start_act, total_duration_ms)) | |
| # Duck background audio with 80ms crossfades | |
| ducked_bg = bg_audio | |
| crossfade_ms = 80 | |
| fully_ducked_bg = bg_audio - 10 | |
| for start, end in active_intervals: | |
| if end - start <= 0: | |
| continue | |
| if end - start < crossfade_ms * 2: | |
| # If segment is too short, just duck it directly | |
| ducked_bg = ducked_bg[:start] + fully_ducked_bg[start:end] + ducked_bg[end:] | |
| else: | |
| # Crossfade normal -> ducked at the start | |
| normal_fade_in_zone = bg_audio[start:start+crossfade_ms] | |
| ducked_fade_in_zone = fully_ducked_bg[start:start+crossfade_ms] | |
| transition_in = normal_fade_in_zone.fade_out(crossfade_ms).overlay(ducked_fade_in_zone.fade_in(crossfade_ms)) | |
| # Crossfade ducked -> normal at the end | |
| ducked_fade_out_zone = fully_ducked_bg[end-crossfade_ms:end] | |
| normal_fade_out_zone = bg_audio[end-crossfade_ms:end] | |
| transition_out = ducked_fade_out_zone.fade_out(crossfade_ms).overlay(normal_fade_out_zone.fade_in(crossfade_ms)) | |
| # Fully ducked middle part | |
| ducked_middle = fully_ducked_bg[start+crossfade_ms:end-crossfade_ms] | |
| ducked_bg = ducked_bg[:start] + transition_in + ducked_middle + transition_out + ducked_bg[end:] | |
| # Nếu mute original mà rơi vào legacy branch (segment_actions rỗng): vẫn ưu tiên mute | |
| if mute_original_audio: | |
| final_audio = AudioSegment.silent(duration=total_duration_ms) | |
| if vi_voiceover_enabled: | |
| final_audio = final_audio.overlay(voice_overlays) | |
| output_path.parent.mkdir(parents=True, exist_ok=True) | |
| final_audio.export(str(output_path), format="wav") | |
| print(f"[MUTE ORIGINAL] Legacy mute exported to {output_path}") | |
| return True | |
| # Merge vocal segments in preserve regions (English dialogue) | |
| final_audio = ducked_bg | |
| for start, end in preserve_intervals: | |
| if start >= total_duration_ms: | |
| continue | |
| clip = vocals_audio[start:end] | |
| # Overlay preserved original vocals | |
| final_audio = final_audio.overlay(clip, position=start) | |
| # Overlay Vietnamese TTS voices | |
| if vi_voiceover_enabled: | |
| final_audio = final_audio.overlay(voice_overlays) | |
| # Export final WAV | |
| output_path.parent.mkdir(parents=True, exist_ok=True) | |
| final_audio.export(str(output_path), format="wav") | |
| print(f"Mixed track exported successfully to {output_path}") | |
| return True | |
| def _mix_language_aware(self, original_audio, background_audio, vocals_audio, voice_overlays, segment_actions, chinese_vocal_handling, chinese_vocal_volume_percent, total_duration_ms): | |
| final_audio = original_audio | |
| mode = (chinese_vocal_handling or "lower").strip().lower() | |
| volume_ratio = max(0.0, min(1.0, float(chinese_vocal_volume_percent) / 100.0)) | |
| lower_gain_db = -60.0 if volume_ratio <= 0 else 20.0 * math.log10(volume_ratio) | |
| edited_regions = [] | |
| for row in segment_actions: | |
| # apply_chinese_vocal_control is set True for whichever source language is active (zh or en) | |
| if not row.get("apply_chinese_vocal_control", False): | |
| continue | |
| start = max(0, int(row.get("start_ms", 0))) | |
| end = min(total_duration_ms, int(row.get("end_ms", start))) | |
| if end <= start: | |
| continue | |
| if mode == "keep": | |
| edited_regions.append({"id": row.get("id"), "start_ms": start, "end_ms": end, "mode": "keep"}) | |
| continue | |
| if mode in ("mute", "remove_if_possible"): | |
| replacement = background_audio[start:end] | |
| if len(replacement) <= 0: | |
| replacement = original_audio[start:end] - 35 | |
| applied_mode = "remove_vocal_stem" if mode == "remove_if_possible" else "mute_to_background" | |
| else: | |
| bg_clip = background_audio[start:end] | |
| vocal_clip = vocals_audio[start:end] + lower_gain_db | |
| replacement = bg_clip.overlay(vocal_clip) | |
| applied_mode = f"lower_to_{int(volume_ratio * 100)}pct" | |
| replacement = self._fit_duration(replacement, end - start) | |
| final_audio = final_audio[:start] + replacement + final_audio[end:] | |
| edited_regions.append({"id": row.get("id"), "start_ms": start, "end_ms": end, "mode": applied_mode}) | |
| final_audio = final_audio.overlay(voice_overlays) | |
| print(f"Language-aware Chinese vocal regions edited: {len(edited_regions)}") | |
| return final_audio | |
| def _fit_duration(self, audio, target_ms): | |
| if len(audio) == target_ms: | |
| return audio | |
| if len(audio) > target_ms: | |
| return audio[:target_ms] | |
| return audio + AudioSegment.silent(duration=target_ms - len(audio)) | |
| def _gain_db_from_percent(self, percent): | |
| ratio = max(0.0, float(percent) / 100.0) | |
| if ratio <= 0: | |
| return None | |
| return 20.0 * math.log10(ratio) | |