fix(reels): garder le son d'origine de la vidéo quand aucune musique n'est choisie

Sans musique (bouton « Passer »), la bande son de la vidéo était supprimée
dès que la voix était activée. Elle sert désormais de fond sonore : conservée
telle quelle, baissée automatiquement sous la voix, puis normalisée avec le
reste. Une musique choisie la remplace toujours.

Même comportement dans l'aperçu, le rendu Remotion et le rendu FFmpeg de
secours. Le bouton indique « Passer (garder le son de la vidéo) ».

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_018Ze4bs7tpF1KGWUk6ZZSZ4
This commit is contained in:
Claude committed 2026-09-28 08:57:38 +00:00
1 parent 1d47763b29
commit f3113a0f5a
4 files changed
+91 -38

No files matched your search

+2 -2
View File
@@ -694,7 +694,7 @@ export default function NewReel() {
setSelectedTrack(null);
setCurrentStep('text');
}}>
Passer
Passer (garder le son de la vidéo)
</Button>
<Button onClick={() => setCurrentStep('text')}>
Continuer
@@ -1052,7 +1052,7 @@ export default function NewReel() {
</div>
<div className="flex justify-between">
<span className="text-muted-foreground">Musique</span>
<span>{selectedTrack ? selectedTrack.title : 'Aucune'}</span>
<span>{selectedTrack ? selectedTrack.title : 'Son de la vidéo'}</span>
</div>
<div className="flex justify-between">
<span className="text-muted-foreground">Texte</span>
+10 -3
View File
@@ -36,9 +36,16 @@ export const ReelVideo: React.FC<ReelVideoProps> = (props) => {
const videoFrames = Math.max(1, Math.floor(videoDuration * fps));
const totalFrames = Math.max(1, Math.round(totalDuration * fps));
const hasAddedAudio = Boolean(mixedAudioUrl || voiceUrl || musicUrl);
// Le son d'origine est remplacé par la musique choisie ; sans musique, il reste
// en fond (baissé sous la voix). Au rendu final, il est déjà dans la piste mixée.
const keepVideoSound = !mixedAudioUrl && !musicUrl;
const video = (
<OffthreadVideo src={videoUrl} muted={hasAddedAudio} style={{ width: "100%", height: "100%", objectFit: "cover" }} />
<OffthreadVideo
src={videoUrl}
muted={!keepVideoSound}
volume={keepVideoSound && voiceUrl ? (frame) => previewMusicVolume(frame / fps, words, 1, fadeStart) : 1}
style={{ width: "100%", height: "100%", objectFit: "cover" }}
/>
);
return (
@@ -78,7 +85,7 @@ export const ReelVideo: React.FC<ReelVideoProps> = (props) => {
);
};
/** Aperçu : la musique baisse pendant la parole et s'éteint avec le fondu final. */
/** Aperçu : le fond sonore (musique ou son d'origine) baisse pendant la parole et s'éteint avec le fondu final. */
function previewMusicVolume(time: number, words: TimedWord[], base: number, fadeStart: number | null): number {
const speaking = words.some((w) => time >= w.start - 0.2 && time <= w.end + 0.3);
const fade =
+35 -32
View File
@@ -102,28 +102,41 @@ def video_filters(plan: RenderPlan) -> list[str]:
return chain
def uses_original_audio(plan: RenderPlan) -> bool:
"""Sans musique choisie, la bande son d'origine de la vidéo sert de fond sonore."""
return plan.music is None and plan.keep_original_audio
def audio_graph(
plan: RenderPlan, music_idx: int | None, voice_idx: int | None
plan: RenderPlan, music_idx: int | None, voice_idx: int | None, original_idx: int | None = None
) -> tuple[list[str], str | None]:
"""Graphe audio : voix décalée, musique bouclée et baissée sous la voix,
niveau final ~ -14 LUFS. Renvoie les filtres et l'étiquette de sortie."""
"""Graphe audio : voix décalée sur un fond sonore (musique bouclée, ou à
défaut bande son d'origine) baissé sous la voix, niveau final ~ -14 LUFS.
Renvoie les filtres et l'étiquette de sortie."""
graph: list[str] = []
if voice_idx is not None:
delay_ms = int(plan.voice_delay * 1000)
graph.append(f"[{voice_idx}:a]aresample=48000,adelay={delay_ms}:all=1,apad[voice]")
if music_idx is not None:
graph.append(f"[{music_idx}:a]aresample=48000,volume={plan.music_volume:.3f}[music]")
if voice_idx is not None and music_idx is not None:
# La musique baisse automatiquement quand la voix parle (ducking)
bed = None
if music_idx is not None:
graph.append(f"[{music_idx}:a]aresample=48000,volume={plan.music_volume:.3f}[bed]")
bed = "bed"
elif original_idx is not None:
# Son d'origine conservé à son niveau ; il se tait après la fin de la vidéo
graph.append(f"[{original_idx}:a:0]aresample=48000,apad[bed]")
bed = "bed"
if voice_idx is not None and bed:
# Le fond sonore baisse automatiquement quand la voix parle (ducking)
graph.append("[voice]asplit=2[vmix][vkey]")
graph.append("[music][vkey]sidechaincompress=threshold=0.02:ratio=8:attack=20:release=400[ducked]")
graph.append(f"[{bed}][vkey]sidechaincompress=threshold=0.02:ratio=8:attack=20:release=400[ducked]")
graph.append("[ducked][vmix]amix=inputs=2:duration=longest:normalize=0[mix]")
mix = "mix"
elif voice_idx is not None:
mix = "voice"
elif music_idx is not None:
mix = "music"
elif bed:
mix = bed
else:
return graph, None
@@ -206,14 +219,11 @@ def build_command(plan: RenderPlan) -> list[str]:
tail.append("format=yuv420p")
graph.append(f"[{current}]{','.join(tail)}[vout]")
audio_filters, audio_out = audio_graph(plan, music_idx, voice_idx)
# La vidéo (entrée 0) fournit la bande son d'origine si aucune musique n'est choisie
original_idx = 0 if uses_original_audio(plan) else None
audio_filters, audio_out = audio_graph(plan, music_idx, voice_idx, original_idx)
graph += audio_filters
if audio_out:
audio_map = ["-map", f"[{audio_out}]"]
elif plan.keep_original_audio:
audio_map = ["-map", "0:a:0"]
else:
audio_map = []
audio_map = ["-map", f"[{audio_out}]"] if audio_out else []
cmd += ["-filter_complex", ";".join(graph), "-map", "[vout]", *audio_map]
cmd += ["-t", f"{total:.3f}", *_video_encoding(19, "medium")]
@@ -236,21 +246,14 @@ def build_prepared_video_command(plan: RenderPlan, output: Path) -> list[str]:
def build_audio_mix_command(plan: RenderPlan, output: Path) -> list[str] | None:
"""Piste son finale (WAV 48 kHz stéréo), ou None s'il n'y a aucun son à produire."""
total = f"{plan.total_duration:.3f}"
if not plan.music and not plan.voice:
if not plan.keep_original_audio:
return None
# Son d'origine seul : même niveau final que les autres Reels
chain = ["aresample=48000", "loudnorm=I=-14:TP=-1.5:LRA=11", "apad"]
if plan.ending_effect:
chain.append(f"afade=t=out:st={plan.fade_start:.3f}:d={FADE_SECONDS}")
return [
"ffmpeg", "-y", "-hide_banner", "-loglevel", "error", "-i", str(plan.video),
"-map", "0:a:0", "-af", ",".join(chain), "-t", total,
"-ar", "48000", "-ac", "2", "-c:a", "pcm_s16le", str(output),
] # fmt: skip
audio_args, music_idx, voice_idx, _ = _audio_inputs(plan, 0)
graph, audio_out = audio_graph(plan, music_idx, voice_idx)
audio_args, music_idx, voice_idx, index = _audio_inputs(plan, 0)
original_idx = None
if uses_original_audio(plan):
audio_args += ["-i", str(plan.video)]
original_idx = index
graph, audio_out = audio_graph(plan, music_idx, voice_idx, original_idx)
if not audio_out:
return None
return [
"ffmpeg", "-y", "-hide_banner", "-loglevel", "error", *audio_args,
"-filter_complex", ";".join(graph), "-map", f"[{audio_out}]", "-t", total,
+44 -1
View File
@@ -51,7 +51,50 @@ def test_original_audio_kept_when_nothing_added():
video=Path("in.mp4"), video_duration=5.0, output=Path("out.mp4"), keep_original_audio=True
)
cmd = build_command(plan)
assert cmd[cmd.index("-map", cmd.index("[vout]")) + 1] == "0:a:0"
assert "[0:a:0]aresample=48000,apad[bed]" in _graph(cmd)
assert cmd[cmd.index("-map", cmd.index("[vout]")) + 1] == "[aout]"
def test_original_audio_stays_under_the_voice_when_no_music_is_chosen():
plan = RenderPlan(
video=Path("in.mp4"),
video_duration=10.0,
output=Path("out.mp4"),
voice=Path("v.wav"),
voice_duration=4.0,
keep_original_audio=True,
)
graph = _graph(build_command(plan))
assert "[0:a:0]aresample=48000,apad[bed]" in graph
assert "[bed][vkey]sidechaincompress" in graph
def test_chosen_music_replaces_the_original_audio():
plan = RenderPlan(
video=Path("in.mp4"),
video_duration=10.0,
output=Path("out.mp4"),
music=Path("m.mp3"),
keep_original_audio=True,
)
assert "0:a:0" not in _graph(build_command(plan))
def test_prepared_audio_mix_keeps_original_audio_under_voice():
from app.render import build_audio_mix_command
plan = RenderPlan(
video=Path("in.mp4"),
video_duration=10.0,
output=Path("out.mp4"),
voice=Path("v.wav"),
voice_duration=4.0,
keep_original_audio=True,
)
cmd = build_audio_mix_command(plan, Path("a.wav"))
# Entrées : voix (0) puis vidéo d'origine (1)
assert cmd[cmd.index("in.mp4") - 1] == "-i"
assert "[1:a:0]aresample=48000,apad[bed]" in " ".join(cmd)
def test_encoding_targets_social_networks():