mirror of
https://github.com/R0m1k3/Socialflow.git
synced 2026-10-11 17:26:45 +02:00
fix(voix): plus de fins de phrases coupées ni de bruits parasites
Fins de phrases coupées : - Le fondu final ne commence plus avant la fin de la voix (il rognait la dernière phrase quand la vidéo finissait peu après) ; il se raccourcit au besoin. Même règle côté Python, aperçu et rendu Remotion - SRT : découpe au milieu de la zone la plus calme de chaque pause, mesurée sur l'enveloppe sonore (tranches de 10 ms), avec un seuil relatif au niveau de la voix qui respecte les fins douces ; 300 ms gardées après le dernier son Bruits parasites : - Normalisation de la voix en deux passes, en mode linéaire : le mode dynamique relevait souffles et bruits entre les phrases (mesuré : -64 dB avant, -75 dB après dans les pauses) - Réduction de bruit douce (afftdn), compresseur sans gain ajouté, fondus de 5 ms aux extrémités ; mix final moins « pompant » (LRA 20) - Voix Gemini vérifiée : les bruits ajoutés avant le premier ou après le dernier mot sont retirés Vérification de lecture comparée lettre à lettre : les petites erreurs de Whisper ne déclenchent plus de régénérations inutiles. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_018Ze4bs7tpF1KGWUk6ZZSZ4
This commit is contained in:
16 files changed
+338
-152
No files matched your search
@@ -144,6 +144,7 @@ export function ReelPreview(props: ReelPreviewProps) {
|
||||
storeName,
|
||||
logoStart: timing.logoStart,
|
||||
fadeStart: timing.fadeStart,
|
||||
fadeDuration: timing.fadeDuration,
|
||||
voiceUrl: srtCues ? undefined : voice?.audioUrl,
|
||||
voiceDelay: VOICE_DELAY,
|
||||
musicUrl,
|
||||
|
||||
@@ -17,6 +17,7 @@ export type ReelVideoProps = {
|
||||
storeName?: string;
|
||||
logoStart?: number | null;
|
||||
fadeStart?: number | null;
|
||||
fadeDuration?: number | null;
|
||||
/** Rendu final : piste son déjà mixée (voix, musique baissée sous la voix, niveau normalisé). */
|
||||
mixedAudioUrl?: string;
|
||||
/** Aperçu : pistes séparées, mixées dans le navigateur. */
|
||||
@@ -31,7 +32,7 @@ export const ReelVideo: React.FC<ReelVideoProps> = (props) => {
|
||||
const { fps } = useVideoConfig();
|
||||
const {
|
||||
videoUrl, videoDuration, totalDuration, words, captionStyle, logoUrl, showWatermark = true, storeName,
|
||||
logoStart = null, fadeStart = null, mixedAudioUrl, voiceUrl, voiceDelay = 2, musicUrl, musicVolume = 0.25,
|
||||
logoStart = null, fadeStart = null, fadeDuration = FADE_SECONDS, mixedAudioUrl, voiceUrl, voiceDelay = 2, musicUrl, musicVolume = 0.25,
|
||||
} = props;
|
||||
|
||||
const videoFrames = Math.max(1, Math.floor(videoDuration * fps));
|
||||
@@ -43,7 +44,7 @@ export const ReelVideo: React.FC<ReelVideoProps> = (props) => {
|
||||
<OffthreadVideo
|
||||
src={videoUrl}
|
||||
muted={!keepVideoSound}
|
||||
volume={keepVideoSound && voiceUrl ? (frame) => previewMusicVolume(frame / fps, words, 1, fadeStart) : 1}
|
||||
volume={keepVideoSound && voiceUrl ? (frame) => previewMusicVolume(frame / fps, words, 1, fadeStart, fadeDuration ?? FADE_SECONDS) : 1}
|
||||
style={{ width: "100%", height: "100%", objectFit: "cover" }}
|
||||
/>
|
||||
);
|
||||
@@ -71,7 +72,7 @@ export const ReelVideo: React.FC<ReelVideoProps> = (props) => {
|
||||
<Html5Audio
|
||||
src={musicUrl}
|
||||
loop
|
||||
volume={(frame) => previewMusicVolume(frame / fps, words, musicVolume, fadeStart)}
|
||||
volume={(frame) => previewMusicVolume(frame / fps, words, musicVolume, fadeStart, fadeDuration ?? FADE_SECONDS)}
|
||||
/>
|
||||
)}
|
||||
|
||||
@@ -80,18 +81,24 @@ export const ReelVideo: React.FC<ReelVideoProps> = (props) => {
|
||||
{logoStart != null && (logoUrl || storeName) && (
|
||||
<Outro start={logoStart} logoUrl={logoUrl} storeName={storeName} />
|
||||
)}
|
||||
{fadeStart != null && <FadeOut start={fadeStart} duration={FADE_SECONDS} />}
|
||||
{fadeStart != null && <FadeOut start={fadeStart} duration={fadeDuration ?? FADE_SECONDS} />}
|
||||
</AbsoluteFill>
|
||||
);
|
||||
};
|
||||
|
||||
/** Aperçu : le fond sonore (musique ou son d'origine) baisse pendant la parole et s'éteint avec le fondu final. */
|
||||
function previewMusicVolume(time: number, words: TimedWord[], base: number, fadeStart: number | null): number {
|
||||
function previewMusicVolume(
|
||||
time: number,
|
||||
words: TimedWord[],
|
||||
base: number,
|
||||
fadeStart: number | null,
|
||||
fadeDuration: number,
|
||||
): number {
|
||||
const speaking = words.some((w) => time >= w.start - 0.2 && time <= w.end + 0.3);
|
||||
const fade =
|
||||
fadeStart == null
|
||||
? 1
|
||||
: interpolate(time, [fadeStart, fadeStart + FADE_SECONDS], [1, 0], {
|
||||
: interpolate(time, [fadeStart, fadeStart + fadeDuration], [1, 0], {
|
||||
extrapolateLeft: "clamp",
|
||||
extrapolateRight: "clamp",
|
||||
});
|
||||
|
||||
@@ -407,6 +407,8 @@ async def _prepare(request: ReelRequest, job_id: str, workdir: Path, started: fl
|
||||
"total_duration": plan.total_duration,
|
||||
"video_duration": info.duration,
|
||||
"logo_start": plan.logo_start if plan.has_outro else None,
|
||||
"fade_start": plan.fade_start if plan.ending_effect else None,
|
||||
"fade_duration": plan.fade_duration if plan.ending_effect else None,
|
||||
"words": [w.to_dict() for w in words],
|
||||
**_voice_info(track),
|
||||
"processing_stats": clock.summary(),
|
||||
|
||||
+59
-25
@@ -1,42 +1,76 @@
|
||||
"""Traitement de la voix : filtrage, compression et niveau sonore constant."""
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
from . import proc
|
||||
|
||||
# Coupe les basses inutiles, lisse la dynamique et amène la voix à -16 LUFS :
|
||||
# elle reste intelligible par-dessus la musique sans saturer.
|
||||
VOICE_CHAIN = (
|
||||
# Nettoyage de la voix : coupe les basses inutiles, retire le souffle de fond
|
||||
# (réduction douce, sans gate qui hacherait les fins de phrases) et arrondit
|
||||
# les pics. Pas de gain ajouté : c'est la normalisation qui règle le niveau.
|
||||
CLEAN_CHAIN = (
|
||||
"highpass=f=80,"
|
||||
"acompressor=threshold=0.1:ratio=3:attack=5:release=120:makeup=2,"
|
||||
"loudnorm=I=-16:TP=-1.5:LRA=7,"
|
||||
"aresample=48000"
|
||||
"afftdn=nr=10:nf=-50:tn=1,"
|
||||
"acompressor=threshold=0.125:ratio=2.5:attack=5:release=150:makeup=1"
|
||||
)
|
||||
# Niveau cible de la voix : intelligible par-dessus la musique sans saturer
|
||||
TARGET = "I=-16:TP=-1.5:LRA=11"
|
||||
# Fondu de 5 ms aux extrémités : pas de clic au début ni à la fin
|
||||
EDGE_FADE = 0.005
|
||||
|
||||
|
||||
async def process_voice(source: Path, target: Path) -> None:
|
||||
"""Produit un WAV 48 kHz mono prêt à mixer."""
|
||||
await proc.run(
|
||||
"""Produit un WAV 48 kHz mono prêt à mixer.
|
||||
|
||||
Normalisation en deux passes, en mode linéaire : un seul gain appliqué à
|
||||
toute la voix. Le mode dynamique (une passe) relevait les passages calmes
|
||||
et amplifiait souffles et bruits entre les phrases."""
|
||||
measure = await proc.run(
|
||||
[
|
||||
"ffmpeg",
|
||||
"-y",
|
||||
"-hide_banner",
|
||||
"-loglevel",
|
||||
"error",
|
||||
"-i",
|
||||
str(source),
|
||||
"-af",
|
||||
VOICE_CHAIN,
|
||||
"-ac",
|
||||
"1",
|
||||
"-ar",
|
||||
"48000",
|
||||
"-c:a",
|
||||
"pcm_s16le",
|
||||
str(target),
|
||||
"ffmpeg", "-hide_banner", "-nostats", "-i", str(source),
|
||||
"-af", f"{CLEAN_CHAIN},loudnorm={TARGET}:print_format=json", "-f", "null", "-",
|
||||
],
|
||||
timeout=120,
|
||||
)
|
||||
stderr_output=True,
|
||||
) # fmt: skip
|
||||
stats = _loudnorm_stats(measure)
|
||||
if stats:
|
||||
loudnorm = (
|
||||
f"loudnorm={TARGET}:linear=true:"
|
||||
f"measured_I={stats['input_i']}:measured_TP={stats['input_tp']}:"
|
||||
f"measured_LRA={stats['input_lra']}:measured_thresh={stats['input_thresh']}:"
|
||||
f"offset={stats['target_offset']}"
|
||||
)
|
||||
else:
|
||||
loudnorm = f"loudnorm={TARGET}"
|
||||
duration = (await proc.probe(source)).duration
|
||||
fades = f"afade=t=in:d={EDGE_FADE},afade=t=out:st={max(0.0, duration - EDGE_FADE):.3f}:d={EDGE_FADE}"
|
||||
await proc.run(
|
||||
[
|
||||
"ffmpeg", "-y", "-hide_banner", "-loglevel", "error", "-i", str(source),
|
||||
"-af", f"{CLEAN_CHAIN},{loudnorm},aresample=48000,{fades}",
|
||||
"-ac", "1", "-ar", "48000", "-c:a", "pcm_s16le", str(target),
|
||||
],
|
||||
timeout=120,
|
||||
) # fmt: skip
|
||||
|
||||
|
||||
def _loudnorm_stats(ffmpeg_log: str) -> dict[str, str] | None:
|
||||
"""Mesures de la première passe (bloc JSON écrit par loudnorm)."""
|
||||
start, end = ffmpeg_log.rfind("{"), ffmpeg_log.rfind("}")
|
||||
if start < 0 or end <= start:
|
||||
return None
|
||||
try:
|
||||
stats = json.loads(ffmpeg_log[start : end + 1])
|
||||
except json.JSONDecodeError:
|
||||
return None
|
||||
keys = ("input_i", "input_tp", "input_lra", "input_thresh", "target_offset")
|
||||
if not all(k in stats for k in keys):
|
||||
return None
|
||||
# Voix quasi muette : les mesures valent -inf, la normalisation simple suffit
|
||||
if any("inf" in str(stats[k]) for k in keys):
|
||||
return None
|
||||
return {k: stats[k] for k in keys}
|
||||
|
||||
|
||||
async def encode_preview(source: Path, target: Path) -> None:
|
||||
|
||||
+104
-63
@@ -2,23 +2,21 @@
|
||||
|
||||
Gemini lit parfois mal : mots sautés, fin tronquée, consigne de style lue à
|
||||
voix haute. On compare donc ce que Whisper entend au texte attendu, et on
|
||||
découpe la voix dans ses vrais silences (mesurés dans l'audio) plutôt qu'aux
|
||||
bornes des mots estimées par Whisper, trop imprécises pour ne pas rogner une
|
||||
syllabe.
|
||||
découpe la voix d'après son niveau sonore réel (mesuré toutes les 10 ms)
|
||||
plutôt qu'aux bornes des mots estimées par Whisper, trop imprécises pour ne
|
||||
pas rogner une syllabe.
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import difflib
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
|
||||
from . import proc
|
||||
from .align import Word, normalize
|
||||
|
||||
# Seuils de silence : en dessous de -40 dB pendant au moins 120 ms
|
||||
SILENCE_NOISE_DB = -40
|
||||
SILENCE_MIN_SECONDS = 0.12
|
||||
|
||||
|
||||
@dataclass
|
||||
class ReadingCheck:
|
||||
@@ -40,9 +38,12 @@ class ReadingCheck:
|
||||
|
||||
|
||||
def check_reading(expected_text: str, spoken: list[Word]) -> ReadingCheck:
|
||||
"""Compare le texte attendu à ce qui a été entendu (Whisper)."""
|
||||
expected = [t for t in (normalize(w) for w in expected_text.split()) if t]
|
||||
heard = [t for t in (normalize(w.text) for w in spoken) if t]
|
||||
"""Compare le texte attendu à ce qui a été entendu (Whisper), lettre à
|
||||
lettre : les petites erreurs de reconnaissance (« petit » pour « petits »,
|
||||
« week -end » découpé) ne comptent pas comme des mots manquants, alors
|
||||
qu'un mot sauté, une fin tronquée ou une phrase ajoutée se voient."""
|
||||
expected = "".join(normalize(w) for w in expected_text.split())
|
||||
heard = "".join(normalize(w.text) for w in spoken)
|
||||
if not expected:
|
||||
return ReadingCheck(1.0, 0.0, True)
|
||||
if not heard:
|
||||
@@ -51,68 +52,108 @@ def check_reading(expected_text: str, spoken: list[Word]) -> ReadingCheck:
|
||||
matcher = difflib.SequenceMatcher(a=expected, b=heard, autojunk=False)
|
||||
blocks = [b for b in matcher.get_matching_blocks() if b.size]
|
||||
matched = sum(b.size for b in blocks)
|
||||
# Fin lue : un des deux derniers mots attendus est reconnu
|
||||
tail = {len(expected) - 1, len(expected) - 2}
|
||||
ending_ok = any(b.a <= i < b.a + b.size for b in blocks for i in tail if i >= 0)
|
||||
# Fin lue : une bonne partie des dernières lettres du texte est reconnue
|
||||
tail_start = max(0, len(expected) - 8)
|
||||
tail_matched = sum(max(0, min(b.a + b.size, len(expected)) - max(b.a, tail_start)) for b in blocks)
|
||||
return ReadingCheck(
|
||||
coverage=matched / len(expected),
|
||||
extra=max(0, len(heard) - matched) / len(expected),
|
||||
ending_ok=ending_ok,
|
||||
ending_ok=tail_matched >= min(4, len(expected) - tail_start),
|
||||
)
|
||||
|
||||
|
||||
_SILENCE_START = re.compile(r"silence_start: (-?[\d.]+)")
|
||||
_SILENCE_END = re.compile(r"silence_end: (-?[\d.]+)")
|
||||
# Enveloppe sonore : niveau de la voix toutes les 10 ms
|
||||
FRAME = 0.01
|
||||
# Est « audible » ce qui dépasse le niveau de crête moins 38 dB : un seuil
|
||||
# relatif respecte les fins de phrases douces (« s » final, voix qui retombe),
|
||||
# qu'un seuil absolu prenait pour du silence.
|
||||
ACTIVE_RANGE_DB = 38.0
|
||||
# Marges gardées autour de la parole
|
||||
LEAD_PAD = 0.08
|
||||
TAIL_PAD = 0.30
|
||||
|
||||
|
||||
def parse_silences(ffmpeg_log: str, duration: float) -> list[tuple[float, float]]:
|
||||
"""Intervalles de silence tirés de la sortie du filtre silencedetect."""
|
||||
silences: list[tuple[float, float]] = []
|
||||
start: float | None = None
|
||||
for line in ffmpeg_log.splitlines():
|
||||
if m := _SILENCE_START.search(line):
|
||||
start = max(0.0, float(m.group(1)))
|
||||
elif (m := _SILENCE_END.search(line)) and start is not None:
|
||||
silences.append((start, float(m.group(1))))
|
||||
start = None
|
||||
if start is not None:
|
||||
silences.append((start, duration))
|
||||
return silences
|
||||
@dataclass
|
||||
class Envelope:
|
||||
db: np.ndarray # niveau (dBFS) de chaque tranche de 10 ms
|
||||
|
||||
@property
|
||||
def duration(self) -> float:
|
||||
return len(self.db) * FRAME
|
||||
|
||||
@property
|
||||
def threshold(self) -> float:
|
||||
if not len(self.db):
|
||||
return 0.0
|
||||
return float(np.percentile(self.db, 95)) - ACTIVE_RANGE_DB
|
||||
|
||||
def _frames(self, start: float, end: float) -> tuple[int, int]:
|
||||
first = max(0, int(start / FRAME))
|
||||
last = min(len(self.db), int(np.ceil(end / FRAME)))
|
||||
return first, max(first, last)
|
||||
|
||||
def quietest(self, start: float, end: float) -> float:
|
||||
"""Milieu de la zone la plus calme d'un intervalle : la coupe tombe au
|
||||
cœur de la pause, loin des deux phrases qu'elle sépare."""
|
||||
first, last = self._frames(start, end)
|
||||
if last - first < 1:
|
||||
return (start + end) / 2
|
||||
smooth = np.convolve(self.db, np.ones(5) / 5, mode="same")[first:last]
|
||||
quiet = smooth <= smooth.min() + 3.0 # à 3 dB du minimum
|
||||
best_start, best_length, run_start = 0, 0, None
|
||||
for i, is_quiet in enumerate([*quiet, False]):
|
||||
if is_quiet and run_start is None:
|
||||
run_start = i
|
||||
elif not is_quiet and run_start is not None:
|
||||
if i - run_start > best_length:
|
||||
best_start, best_length = run_start, i - run_start
|
||||
run_start = None
|
||||
return (first + best_start + best_length / 2) * FRAME
|
||||
|
||||
def first_sound(self, start: float, end: float) -> float | None:
|
||||
first, last = self._frames(start, end)
|
||||
active = np.nonzero(self.db[first:last] > self.threshold)[0]
|
||||
return (first + int(active[0])) * FRAME if len(active) else None
|
||||
|
||||
def last_sound(self, start: float, end: float) -> float | None:
|
||||
first, last = self._frames(start, end)
|
||||
active = np.nonzero(self.db[first:last] > self.threshold)[0]
|
||||
return (first + int(active[-1]) + 1) * FRAME if len(active) else None
|
||||
|
||||
def speech_span(self, start: float, end: float) -> tuple[float, float]:
|
||||
"""Parole contenue dans [start, end], avec ses marges, sans en sortir."""
|
||||
onset, offset = self.first_sound(start, end), self.last_sound(start, end)
|
||||
if onset is None or offset is None:
|
||||
return start, end
|
||||
return max(start, onset - LEAD_PAD), min(end, offset + TAIL_PAD)
|
||||
|
||||
|
||||
async def detect_silences(path: Path, duration: float) -> list[tuple[float, float]]:
|
||||
log = await proc.run(
|
||||
[
|
||||
"ffmpeg", "-hide_banner", "-nostats", "-i", str(path),
|
||||
"-af", f"silencedetect=noise={SILENCE_NOISE_DB}dB:d={SILENCE_MIN_SECONDS}",
|
||||
"-f", "null", "-",
|
||||
],
|
||||
timeout=120,
|
||||
stderr_output=True,
|
||||
def envelope_from_samples(samples: np.ndarray, rate: int) -> Envelope:
|
||||
per_frame = int(rate * FRAME)
|
||||
count = len(samples) // per_frame
|
||||
if count == 0:
|
||||
return Envelope(np.array([-120.0]))
|
||||
frames = samples[: count * per_frame].reshape(count, per_frame).astype(np.float64)
|
||||
rms = np.sqrt(np.mean(frames**2, axis=1))
|
||||
return Envelope(20 * np.log10(np.maximum(rms, 1e-6)))
|
||||
|
||||
|
||||
async def load_envelope(path: Path) -> Envelope:
|
||||
"""Décode l'audio (mono 16 kHz) et mesure son niveau toutes les 10 ms."""
|
||||
process = await asyncio.create_subprocess_exec(
|
||||
"ffmpeg", "-hide_banner", "-loglevel", "error", "-i", str(path),
|
||||
"-ac", "1", "-ar", "16000", "-f", "f32le", "-",
|
||||
stdout=asyncio.subprocess.PIPE, stderr=asyncio.subprocess.PIPE,
|
||||
) # fmt: skip
|
||||
return parse_silences(log, duration)
|
||||
raw, err = await process.communicate()
|
||||
if process.returncode != 0:
|
||||
raise proc.CommandError(["ffmpeg"], process.returncode, err.decode(errors="replace"))
|
||||
return envelope_from_samples(np.frombuffer(raw, dtype=np.float32), 16000)
|
||||
|
||||
|
||||
def speech_bounds(
|
||||
silences: list[tuple[float, float]], duration: float, pad: float = 0.08
|
||||
) -> tuple[float, float]:
|
||||
"""Début et fin de la parole (silences d'ouverture et de fin retirés, avec une marge)."""
|
||||
start, end = 0.0, duration
|
||||
for s, e in silences:
|
||||
if s <= 0.01:
|
||||
start = max(start, e)
|
||||
if e >= duration - 0.01:
|
||||
end = min(end, s)
|
||||
start, end = max(0.0, start - pad), min(duration, end + pad)
|
||||
return (start, end) if end > start else (0.0, duration)
|
||||
|
||||
|
||||
def cut_point(silences: list[tuple[float, float]], after: float, before: float) -> float:
|
||||
"""Instant de coupe entre deux phrases : au milieu du plus long silence
|
||||
situé entre la fin de l'une et le début de l'autre (bornes Whisper élargies,
|
||||
car imprécises). Sans silence mesuré, le milieu de l'écart."""
|
||||
low, high = after - 0.3, before + 0.3
|
||||
candidates = [(e - s, (s + e) / 2) for s, e in silences if e > low and s < high]
|
||||
if candidates:
|
||||
return max(candidates)[1]
|
||||
return (after + before) / 2
|
||||
def sentence_cut(envelope: Envelope, previous_end: float, next_start: float) -> float:
|
||||
"""Coupe entre deux phrases, au point le plus calme de la pause. Les bornes
|
||||
Whisper sont imprécises : on cherche un peu autour."""
|
||||
low = max(0.0, min(previous_end, next_start) - 0.1)
|
||||
high = max(previous_end, next_start) + 0.2
|
||||
return envelope.quietest(low, high)
|
||||
@@ -14,6 +14,8 @@ from . import config
|
||||
from .subtitles import filter_path
|
||||
|
||||
FADE_SECONDS = 2.0
|
||||
# Fondu raccourci quand la voix finit tard, mais jamais sec
|
||||
MIN_FADE = 0.3
|
||||
LOGO_SECONDS = 5.0
|
||||
# Silence laissé après la dernière phrase avant la fin de la vidéo
|
||||
VOICE_TAIL = 0.8
|
||||
@@ -76,7 +78,16 @@ class RenderPlan:
|
||||
|
||||
@property
|
||||
def fade_start(self) -> float:
|
||||
return max(0.0, self.total_duration - FADE_SECONDS)
|
||||
"""Le fondu final ne commence jamais avant la fin de la voix : il
|
||||
rognait la dernière phrase quand la vidéo finissait peu après."""
|
||||
start = max(0.0, self.total_duration - FADE_SECONDS)
|
||||
if self.voice:
|
||||
start = max(start, min(self.speech_end + 0.15, self.total_duration - MIN_FADE))
|
||||
return round(start, 3)
|
||||
|
||||
@property
|
||||
def fade_duration(self) -> float:
|
||||
return round(max(MIN_FADE, self.total_duration - self.fade_start), 3)
|
||||
|
||||
@property
|
||||
def freeze_duration(self) -> float:
|
||||
@@ -140,9 +151,9 @@ def audio_graph(
|
||||
else:
|
||||
return graph, None
|
||||
|
||||
tail = ["loudnorm=I=-14:TP=-1.5:LRA=11", "aresample=48000"]
|
||||
tail = ["loudnorm=I=-14:TP=-1.5:LRA=20", "aresample=48000"]
|
||||
if plan.ending_effect:
|
||||
tail.append(f"afade=t=out:st={plan.fade_start:.3f}:d={FADE_SECONDS}")
|
||||
tail.append(f"afade=t=out:st={plan.fade_start:.3f}:d={plan.fade_duration}")
|
||||
graph.append(f"[{mix}]{','.join(tail)}[aout]")
|
||||
return graph, "aout"
|
||||
|
||||
@@ -215,7 +226,7 @@ def build_command(plan: RenderPlan) -> list[str]:
|
||||
|
||||
tail = []
|
||||
if plan.ending_effect:
|
||||
tail.append(f"fade=t=out:st={plan.fade_start:.3f}:d={FADE_SECONDS}")
|
||||
tail.append(f"fade=t=out:st={plan.fade_start:.3f}:d={plan.fade_duration}")
|
||||
tail.append("format=yuv420p")
|
||||
graph.append(f"[{current}]{','.join(tail)}[vout]")
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
"""Voix calée sur un fichier SRT.
|
||||
|
||||
Tout le texte est lu d'une seule traite, puis la voix est découpée phrase par
|
||||
phrase dans ses vrais silences et chaque morceau est posé à l'instant de son
|
||||
phrase au point le plus calme de chaque pause, et chaque morceau est posé à l'instant de son
|
||||
sous-titre. Une lecture unique garde la même intonation, le même rythme et le
|
||||
même niveau d'un bout à l'autre ; lire chaque sous-titre séparément donnait
|
||||
une voix différente à chaque phrase, et le découpage aux bornes estimées par
|
||||
@@ -9,7 +9,7 @@ Whisper rognait des syllabes.
|
||||
|
||||
Si la lecture unique ne peut pas être répartie entre les sous-titres, on lit
|
||||
chaque sous-titre séparément (même voix, même graine), en découpant là aussi
|
||||
dans les silences mesurés.
|
||||
au niveau sonore mesuré.
|
||||
|
||||
Quand un morceau est plus long que son sous-titre, il est accéléré (sans
|
||||
changer la hauteur de la voix) pour finir à temps.
|
||||
@@ -28,9 +28,6 @@ log = logging.getLogger(__name__)
|
||||
|
||||
# Débordement toléré sur le silence qui suit un sous-titre avant d'accélérer
|
||||
OVERFLOW_TOLERANCE = 0.25
|
||||
# Marges gardées autour de la parole quand on retire les silences
|
||||
LEAD_PAD = 0.08
|
||||
TAIL_PAD = 0.15
|
||||
# Au-delà, l'accélération s'entend nettement : l'utilisateur est prévenu
|
||||
AUDIBLE_SPEEDUP = 1.35
|
||||
|
||||
@@ -100,29 +97,20 @@ def split_by_cue(words: list[Word], cue_texts: list[str]) -> list[list[Word]] |
|
||||
return groups
|
||||
|
||||
|
||||
def segments_from_reading(
|
||||
groups: list[list[Word]], silences: list[tuple[float, float]], duration: float
|
||||
) -> list[Segment] | None:
|
||||
"""Découpe la lecture complète entre les phrases, au milieu des silences."""
|
||||
start, end = quality.speech_bounds(silences, duration, pad=LEAD_PAD)
|
||||
cuts = [start]
|
||||
def segments_from_reading(groups: list[list[Word]], envelope: quality.Envelope) -> list[Segment] | None:
|
||||
"""Découpe la lecture complète entre les phrases, au point le plus calme de
|
||||
chaque pause, puis retire le silence autour de chaque phrase en gardant une
|
||||
marge (300 ms après le dernier son : les fins de phrases ne sont pas rognées)."""
|
||||
cuts = [0.0]
|
||||
for previous, following in zip(groups, groups[1:]):
|
||||
cuts.append(quality.cut_point(silences, previous[-1].end, following[0].start))
|
||||
cuts.append(end)
|
||||
cuts.append(quality.sentence_cut(envelope, previous[-1].end, following[0].start))
|
||||
cuts.append(envelope.duration)
|
||||
if any(b <= a for a, b in zip(cuts, cuts[1:])):
|
||||
return None
|
||||
|
||||
segments = []
|
||||
for i, group in enumerate(groups):
|
||||
seg_start, seg_end = cuts[i], cuts[i + 1]
|
||||
# Retire le silence de part et d'autre de la coupe, en gardant une marge
|
||||
for s, e in silences:
|
||||
if s <= seg_start < e:
|
||||
seg_start = max(seg_start, e - LEAD_PAD)
|
||||
if s < seg_end <= e:
|
||||
seg_end = min(seg_end, s + TAIL_PAD)
|
||||
if seg_end <= seg_start:
|
||||
seg_start, seg_end = cuts[i], cuts[i + 1]
|
||||
seg_start, seg_end = envelope.speech_span(cuts[i], cuts[i + 1])
|
||||
segments.append(Segment(seg_start, seg_end, group))
|
||||
return segments
|
||||
|
||||
@@ -147,9 +135,9 @@ async def synthesize_cues(
|
||||
# 1. Lecture d'une seule traite, découpée dans les silences
|
||||
script = " ".join(texts)
|
||||
reading = await tts.synthesize(text=script, display_text=script, workdir=workdir, **options)
|
||||
silences = await quality.detect_silences(reading.path, reading.duration)
|
||||
envelope = await quality.load_envelope(reading.path)
|
||||
groups = split_by_cue(reading.words, texts)
|
||||
segments = segments_from_reading(groups, silences, reading.duration) if groups else None
|
||||
segments = segments_from_reading(groups, envelope) if groups else None
|
||||
|
||||
if segments:
|
||||
sources = [(reading.path, seg) for seg in segments]
|
||||
@@ -170,8 +158,8 @@ async def synthesize_cues(
|
||||
**{**options, "engine": used_engine, "voice": used_voice or voice},
|
||||
)
|
||||
warnings += [w for w in track.warnings if w not in warnings]
|
||||
cue_silences = await quality.detect_silences(track.path, track.duration)
|
||||
span = quality.speech_bounds(cue_silences, track.duration, pad=LEAD_PAD)
|
||||
cue_envelope = await quality.load_envelope(track.path)
|
||||
span = cue_envelope.speech_span(0.0, cue_envelope.duration)
|
||||
sources.append((track.path, Segment(span[0], span[1], track.words)))
|
||||
|
||||
placed: list[tuple[Path, float]] = []
|
||||
|
||||
@@ -42,8 +42,10 @@ async def synthesize(
|
||||
else:
|
||||
result = await _gemini_checked(text, voice, style, gemini_api_key, workdir, warnings)
|
||||
if result:
|
||||
processed, spoken, used_voice = result
|
||||
processed, spoken, used_voice, reading_ok = result
|
||||
used_engine = "gemini"
|
||||
if reading_ok:
|
||||
processed, spoken = await _trim_stray_sounds(processed, spoken, workdir)
|
||||
|
||||
if processed is None:
|
||||
raw, spoken, used_voice = await edge.synthesize(text, voice, style, workdir)
|
||||
@@ -61,7 +63,7 @@ async def synthesize(
|
||||
|
||||
async def _gemini_checked(
|
||||
text: str, voice: str | None, style: str | None, api_key: str, workdir: Path, warnings: list[str]
|
||||
) -> tuple[Path, list[Word], str] | None:
|
||||
) -> tuple[Path, list[Word], str, bool] | None:
|
||||
"""Voix Gemini dont la lecture a été vérifiée : Whisper réécoute chaque
|
||||
génération ; mots sautés, fin tronquée ou consigne lue à voix haute
|
||||
déclenchent une nouvelle génération. Renvoie None pour se replier sur Edge."""
|
||||
@@ -97,7 +99,42 @@ async def _gemini_checked(
|
||||
warnings.append(f"Gemini a mal lu le texte ({check.describe()}) : voix Edge utilisée à la place.")
|
||||
return None
|
||||
warnings.append(f"Lecture Gemini imparfaite ({check.describe()}) : écoutez le résultat.")
|
||||
return processed, spoken, gemini_voice
|
||||
return processed, spoken, gemini_voice, check.acceptable
|
||||
|
||||
|
||||
# Au-delà de la fin du dernier mot, ce n'est plus de la parole
|
||||
STRAY_AFTER_LAST_WORD = 0.6
|
||||
STRAY_BEFORE_FIRST_WORD = 0.4
|
||||
|
||||
|
||||
async def _trim_stray_sounds(path: Path, spoken: list[Word], workdir: Path) -> tuple[Path, list[Word]]:
|
||||
"""Retire les bruits que Gemini ajoute parfois avant le premier ou après le
|
||||
dernier mot. Appelé seulement quand la lecture a été vérifiée : le dernier
|
||||
mot est bien reconnu, on ne coupe donc jamais de parole."""
|
||||
if not spoken:
|
||||
return path, spoken
|
||||
envelope = await quality.load_envelope(path)
|
||||
duration = envelope.duration
|
||||
window_start = max(0.0, spoken[0].start - STRAY_BEFORE_FIRST_WORD)
|
||||
window_end = min(duration, spoken[-1].end + STRAY_AFTER_LAST_WORD)
|
||||
start, end = envelope.speech_span(window_start, window_end)
|
||||
if start < 0.02 and end > duration - 0.02:
|
||||
return path, spoken
|
||||
|
||||
trimmed = workdir / "voice_trimmed.wav"
|
||||
fade_out = max(0.0, end - start - 0.01)
|
||||
await proc.run(
|
||||
[
|
||||
"ffmpeg", "-y", "-hide_banner", "-loglevel", "error", "-i", str(path),
|
||||
"-af", f"atrim=start={start:.3f}:end={end:.3f},asetpts=PTS-STARTPTS,"
|
||||
f"afade=t=in:d=0.01,afade=t=out:st={fade_out:.3f}:d=0.01",
|
||||
"-ac", "1", "-ar", "48000", "-c:a", "pcm_s16le", str(trimmed),
|
||||
],
|
||||
timeout=60,
|
||||
) # fmt: skip
|
||||
log.info("Voix recadrée sur la parole : %.2f–%.2f s (sur %.2f s)", start, end, duration)
|
||||
shifted = [Word(w.text, w.start - start, w.end - start) for w in spoken if w.start < end]
|
||||
return trimmed, shifted
|
||||
|
||||
|
||||
def _score(check: quality.ReadingCheck) -> float:
|
||||
|
||||
@@ -5,3 +5,4 @@ httpx==0.28.1
|
||||
edge-tts==7.2.8
|
||||
emoji==2.14.1
|
||||
faster-whisper==1.2.1
|
||||
numpy>=1.26
|
||||
@@ -1,5 +1,7 @@
|
||||
import numpy as np
|
||||
|
||||
from app.align import Word
|
||||
from app.quality import check_reading, cut_point, parse_silences, speech_bounds
|
||||
from app.quality import FRAME, Envelope, check_reading, envelope_from_samples, sentence_cut
|
||||
|
||||
|
||||
def _heard(text: str) -> list[Word]:
|
||||
@@ -23,19 +25,38 @@ def test_instruction_read_aloud_is_rejected():
|
||||
assert not check_reading("Découvrez nos nouveautés en magasin", heard).acceptable
|
||||
|
||||
|
||||
def test_silences_are_parsed_from_ffmpeg_log():
|
||||
log = (
|
||||
"[silencedetect] silence_start: 0\n[silencedetect] silence_end: 0.21 | silence_duration: 0.21\n"
|
||||
"[silencedetect] silence_start: 1.5\n"
|
||||
)
|
||||
assert parse_silences(log, 2.0) == [(0.0, 0.21), (1.5, 2.0)]
|
||||
def _envelope(levels):
|
||||
"""Enveloppe synthétique : liste de (durée en s, niveau en dB)."""
|
||||
db = []
|
||||
for seconds, level in levels:
|
||||
db += [level] * int(round(seconds / FRAME))
|
||||
return Envelope(np.array(db, dtype=float))
|
||||
|
||||
|
||||
def test_speech_bounds_drop_leading_and_trailing_silence():
|
||||
start, end = speech_bounds([(0.0, 0.3), (1.5, 2.0)], 2.0, pad=0.1)
|
||||
assert round(start, 3) == 0.2 and round(end, 3) == 1.6
|
||||
def test_soft_sentence_endings_are_kept():
|
||||
# Parole à -20 dB, fin de phrase douce à -50 dB (relatif : -30 dB, audible),
|
||||
# puis vrai silence à -90 dB
|
||||
env = _envelope([(0.2, -90), (1.0, -20), (0.3, -50), (0.6, -90)])
|
||||
start, end = env.speech_span(0.0, env.duration)
|
||||
assert round(start, 2) == 0.12 # 80 ms avant la parole
|
||||
assert round(end, 2) == 1.8 # 300 ms après la fin douce (à 1,5 s)
|
||||
|
||||
|
||||
def test_cut_point_prefers_the_longest_nearby_silence():
|
||||
assert cut_point([(1.0, 1.1), (1.2, 1.8)], after=1.0, before=1.9) == 1.5
|
||||
assert cut_point([], after=1.0, before=2.0) == 1.5
|
||||
def test_cut_falls_in_the_quietest_part_of_the_pause():
|
||||
env = _envelope([(1.0, -20), (0.15, -60), (0.2, -85), (0.15, -60), (1.0, -20)])
|
||||
cut = sentence_cut(env, previous_end=0.95, next_start=1.55)
|
||||
assert 1.15 <= cut <= 1.35
|
||||
|
||||
|
||||
def test_envelope_from_samples_measures_level():
|
||||
rate = 16000
|
||||
samples = np.concatenate([np.zeros(rate // 2), 0.5 * np.ones(rate // 2)]).astype(np.float32)
|
||||
env = envelope_from_samples(samples, rate)
|
||||
assert env.db[0] < -100 and abs(env.db[-1] - 20 * np.log10(0.5)) < 0.1
|
||||
|
||||
|
||||
def test_small_recognition_errors_are_not_missing_words():
|
||||
expected = "Découvrez nos plantes de saison à petits prix. On vous attend ce week-end."
|
||||
heard = _heard("Découvrez nos plans de saison à petit prix. On vous attend ce week -end.")
|
||||
check = check_reading(expected, heard)
|
||||
assert check.acceptable, check
|
||||
@@ -154,3 +154,18 @@ def test_audio_mix_absent_without_any_sound():
|
||||
assert build_audio_mix_command(plan, Path("a.wav")) is None
|
||||
plan.keep_original_audio = True
|
||||
assert "loudnorm=I=-14" in " ".join(build_audio_mix_command(plan, Path("a.wav")))
|
||||
|
||||
|
||||
def test_final_fade_never_starts_before_the_voice_ends():
|
||||
# Vidéo courte sans logo : 2 s + 5 s de voix, la vidéo finit 0,8 s après.
|
||||
plan = RenderPlan(
|
||||
video=Path("in.mp4"),
|
||||
video_duration=4.0,
|
||||
output=Path("out.mp4"),
|
||||
voice=Path("v.wav"),
|
||||
voice_duration=5.0,
|
||||
)
|
||||
assert plan.total_duration == 7.8
|
||||
assert plan.fade_start >= plan.speech_end
|
||||
assert plan.fade_duration >= 0.3
|
||||
assert f"afade=t=out:st={plan.fade_start:.3f}" in _graph(build_command(plan))
|
||||
@@ -1,6 +1,9 @@
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
|
||||
from app.align import Word, spread_words
|
||||
from app.quality import Envelope
|
||||
from app.render import RenderPlan
|
||||
from app.srt_voice import (
|
||||
Cue,
|
||||
@@ -41,15 +44,16 @@ def test_words_of_the_single_reading_are_split_between_cues():
|
||||
assert split_by_cue(words, ["Bonjour."]) is None
|
||||
|
||||
|
||||
def test_reading_is_cut_inside_silences_not_on_word_bounds():
|
||||
# Whisper situe mal la fin de « Bonjour » (0,9 s) : la vraie parole finit à 1,05 s,
|
||||
# puis silence jusqu'à 1,6 s. La coupe tombe dans ce silence, sans rogner le mot.
|
||||
def test_reading_is_cut_in_the_pause_without_clipping_sentence_endings():
|
||||
# Whisper situe mal la fin de « Bonjour » (0,9 s) : la voix retombe
|
||||
# doucement jusqu'à 1,05 s, puis vraie pause jusqu'à 1,6 s.
|
||||
db = [-90.0] * 15 + [-20.0] * 75 + [-50.0] * 15 + [-90.0] * 55 + [-20.0] * 100 + [-90.0] * 40
|
||||
env = Envelope(np.array(db))
|
||||
groups = [[Word("Bonjour.", 0.2, 0.9)], [Word("Venez", 1.7, 2.1), Word("vite.", 2.1, 2.6)]]
|
||||
silences = [(0.0, 0.15), (1.05, 1.6), (2.7, 3.0)]
|
||||
first, second = segments_from_reading(groups, silences, 3.0)
|
||||
assert first.source_start < 0.15 and 1.05 < first.source_end <= 1.2
|
||||
assert 1.45 <= second.source_start < 1.6 and second.source_end > 2.7
|
||||
assert first.source_end < second.source_start
|
||||
first, second = segments_from_reading(groups, env)
|
||||
assert first.source_end >= 1.05 + 0.25 # fin douce gardée, avec sa marge
|
||||
assert first.source_end <= second.source_start
|
||||
assert 1.5 <= second.source_start < 1.6
|
||||
|
||||
|
||||
def test_each_segment_is_placed_at_its_cue_start():
|
||||
|
||||
@@ -55,6 +55,9 @@ export interface PreparedReel {
|
||||
totalDuration: number;
|
||||
videoDuration: number;
|
||||
logoStart: number | null;
|
||||
/** Fondu final, calculé par le service pour ne jamais rogner la voix. */
|
||||
fadeStart: number | null;
|
||||
fadeDuration: number | null;
|
||||
words: TimedWord[];
|
||||
ttsEngine?: string | null;
|
||||
ttsVoice?: string | null;
|
||||
@@ -226,6 +229,8 @@ export class FFmpegService {
|
||||
total_duration: number;
|
||||
video_duration: number;
|
||||
logo_start: number | null;
|
||||
fade_start?: number | null;
|
||||
fade_duration?: number | null;
|
||||
words: TimedWord[];
|
||||
tts_engine?: string | null;
|
||||
tts_voice?: string | null;
|
||||
@@ -248,6 +253,8 @@ export class FFmpegService {
|
||||
totalDuration: data.total_duration,
|
||||
videoDuration: data.video_duration,
|
||||
logoStart: data.logo_start,
|
||||
fadeStart: data.fade_start ?? null,
|
||||
fadeDuration: data.fade_duration ?? null,
|
||||
words: data.words ?? [],
|
||||
ttsEngine: data.tts_engine,
|
||||
ttsVoice: data.tts_voice,
|
||||
|
||||
@@ -12,7 +12,6 @@
|
||||
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import { FADE_SECONDS } from "@shared/captions";
|
||||
import { videoReelParamsSchema, type VideoReelParams } from "@shared/reel";
|
||||
import { storage } from "../../storage";
|
||||
import { ffmpegService, type ReelRenderOptions } from "../ffmpeg";
|
||||
@@ -139,7 +138,8 @@ async function renderWithRemotion(
|
||||
showWatermark: params.showLogo,
|
||||
storeName: params.enableEndingEffect ? params.storeName : undefined,
|
||||
logoStart: prepared.logoStart,
|
||||
fadeStart: params.enableEndingEffect ? Math.max(0, prepared.totalDuration - FADE_SECONDS) : null,
|
||||
fadeStart: prepared.fadeStart,
|
||||
fadeDuration: prepared.fadeDuration,
|
||||
mixedAudioUrl: prepared.audioPath ? tempFileUrl(prepared.audioPath) : undefined,
|
||||
},
|
||||
onProgress: (ratio) => {
|
||||
|
||||
+10
-1
@@ -24,7 +24,16 @@ describe("computeReelTiming", () => {
|
||||
|
||||
it("n'a ni logo ni fondu sans effet de fin", () => {
|
||||
const timing = computeReelTiming({ videoDuration: 12, hasOutro: true, endingEffect: false });
|
||||
expect(timing).toEqual({ total: 12, logoStart: null, fadeStart: null });
|
||||
expect(timing).toEqual({ total: 12, logoStart: null, fadeStart: null, fadeDuration: null });
|
||||
});
|
||||
});
|
||||
|
||||
describe("fondu final", () => {
|
||||
it("ne commence jamais avant la fin de la voix (mêmes valeurs que Python)", () => {
|
||||
const timing = computeReelTiming({ videoDuration: 4, voiceDuration: 5, hasOutro: false, endingEffect: true });
|
||||
expect(timing.total).toBe(7.8);
|
||||
expect(timing.fadeStart).toBeGreaterThanOrEqual(7);
|
||||
expect(timing.fadeDuration).toBeGreaterThanOrEqual(0.3);
|
||||
});
|
||||
});
|
||||
|
||||
|
||||
+9
-1
@@ -32,12 +32,15 @@ const VOICE_TAIL = 0.8;
|
||||
const OUTRO_MIN = 2.5;
|
||||
const LOGO_SECONDS = 5;
|
||||
export const FADE_SECONDS = 2;
|
||||
/** Fondu raccourci quand la voix finit tard, mais jamais sec. */
|
||||
const MIN_FADE = 0.3;
|
||||
|
||||
export interface ReelTiming {
|
||||
total: number;
|
||||
/** Début de l'effet de fin (grand logo), ou null sans effet de fin. */
|
||||
logoStart: number | null;
|
||||
fadeStart: number | null;
|
||||
fadeDuration: number | null;
|
||||
}
|
||||
|
||||
/** Durée finale : la vidéo s'allonge (dernière image figée) si la voix dépasse. */
|
||||
@@ -57,10 +60,15 @@ export function computeReelTiming(input: {
|
||||
if (hasOutro) total = Math.max(total, speechEnd + OUTRO_MIN);
|
||||
}
|
||||
total = Math.round(total * 1000) / 1000;
|
||||
// Le fondu final ne commence jamais avant la fin de la voix
|
||||
let fadeStart = Math.max(0, total - FADE_SECONDS);
|
||||
if (input.voiceDuration) fadeStart = Math.max(fadeStart, Math.min(speechEnd + 0.15, total - MIN_FADE));
|
||||
fadeStart = Math.round(fadeStart * 1000) / 1000;
|
||||
return {
|
||||
total,
|
||||
logoStart: hasOutro ? Math.max(0, total - LOGO_SECONDS, speechEnd) : null,
|
||||
fadeStart: input.endingEffect ? Math.max(0, total - FADE_SECONDS) : null,
|
||||
fadeStart: input.endingEffect ? fadeStart : null,
|
||||
fadeDuration: input.endingEffect ? Math.max(MIN_FADE, total - fadeStart) : null,
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
Reference in new issue
Block a user