diff --git a/tools/audio/audio_mixer.py b/tools/audio/audio_mixer.py index 44f29499..04a052e3 100644 --- a/tools/audio/audio_mixer.py +++ b/tools/audio/audio_mixer.py @@ -492,17 +492,22 @@ class AudioMixer(BaseTool): duck_enabled = ducking.get("enabled", True) if isinstance(ducking, dict) else bool(ducking) if duck_enabled and speech_tracks and music_tracks: - # Mix speech tracks together first + # Build ONE speech stream, then split it into two independent + # branches: one feeds the sidechain compressor as the ducking key, + # the other is mixed into the final output. A filtergraph label may + # only be consumed once, so reusing the same speech label for both + # the sidechain key and the output mix is invalid on stricter ffmpeg + # builds (e.g. the Linux ffmpeg on CI). asplit makes the fork explicit. speech_indices = list(range(len(speech_tracks))) speech_labels = "".join(f"[a{i}]" for i in speech_indices) if len(speech_tracks) > 1: filter_parts.append( - f"{speech_labels}amix=inputs={len(speech_tracks)}:duration=longest[speech_mix]" + f"{speech_labels}amix=inputs={len(speech_tracks)}:duration=longest[speech_all]" ) - speech_out = "[speech_mix]" else: - speech_out = f"[a{speech_indices[0]}]" + filter_parts.append(f"[a{speech_indices[0]}]acopy[speech_all]") + filter_parts.append("[speech_all]asplit=2[speech_key][speech_out]") # Mix music tracks together music_start = len(speech_tracks) @@ -517,35 +522,20 @@ class AudioMixer(BaseTool): else: music_in = f"[a{music_indices[0]}]" - # Apply sidechain ducking + # Apply sidechain ducking — music is compressed, [speech_key] is the key duck_params = ducking if isinstance(ducking, dict) else {} attack = duck_params.get("attack_ms", 200) / 1000 release = duck_params.get("release_ms", 500) / 1000 music_vol = duck_params.get("music_volume_during_speech", 0.15) filter_parts.append( - f"{music_in}{speech_out}sidechaincompress=" + f"{music_in}[speech_key]sidechaincompress=" f"threshold=0.02:ratio=9:attack={attack}:release={release}:" f"level_sc=1:mix=0.9[ducked_music];" f"[ducked_music]volume={music_vol * 3}[music_out]" ) - # sidechaincompress uses the speech signal only as the ducking key — - # it does not emit speech to the output. Re-derive the speech stream - # for the final mix below. (An earlier version also appended an - # `acopy[speech_dup]` here, but that pad was never consumed and left - # the filtergraph with a dangling output, which ffmpeg rejects — so - # single-narration + music full_mix always failed. FFmpeg auto-splits - # the reused input label, so no explicit duplicate is needed.) - - # Build speech mix for output separately - if len(speech_tracks) > 1: - # speech_mix already exists, make a copy for output - filter_parts.append(f"{speech_labels}amix=inputs={len(speech_tracks)}:duration=longest[speech_out]") - else: - filter_parts.append(f"[a{speech_indices[0]}]acopy[speech_out]") - - # Final mix: speech_out + music_out + # Final mix: the other speech branch + ducked music mix_label = "[speech_out][music_out]amix=inputs=2:duration=longest[premix]" # Add SFX if present