diff --git a/.env.example b/.env.example index 0dad08e..fdcb0ed 100644 --- a/.env.example +++ b/.env.example @@ -120,6 +120,13 @@ GEMINI_API_KEY= # sur celui-ci si le modèle choisi n'existe pas). # GEMINI_TTS_MODEL= # +# Stabilité de la voix Gemini : température basse et graine fixe donnent la +# même intonation d'une génération à l'autre ; chaque lecture est réécoutée et +# régénérée (jusqu'à GEMINI_TTS_ATTEMPTS fois) si des mots manquent. +# GEMINI_TTS_TEMPERATURE=0.3 +# GEMINI_TTS_SEED=20260928 +# GEMINI_TTS_ATTEMPTS=3 +# # Modèle Whisper utilisé pour caler les sous-titres mot à mot sur la voix # (tiny, base, small). Choisi à la construction de l'image : changer la valeur # demande de reconstruire ffmpeg-api. diff --git a/ffmpeg-service/app/align.py b/ffmpeg-service/app/align.py index 4728dbb..0662653 100644 --- a/ffmpeg-service/app/align.py +++ b/ffmpeg-service/app/align.py @@ -116,12 +116,15 @@ def _model(): return WhisperModel(config.WHISPER_MODEL, device="cpu", compute_type=config.WHISPER_COMPUTE_TYPE) -def _transcribe_sync(audio: Path, hint: str | None) -> list[Word]: +def _transcribe_sync(audio: Path) -> list[Word]: + # Pas de texte attendu en amorce (initial_prompt) ni d'enchaînement sur le + # segment précédent : les deux font halluciner Whisper, qui répète la phrase + # et masque une fin tronquée. On veut entendre ce qui est réellement dit. segments, _ = _model().transcribe( str(audio), language="fr", word_timestamps=True, - initial_prompt=hint or None, + condition_on_previous_text=False, vad_filter=False, beam_size=5, ) @@ -133,6 +136,6 @@ def _transcribe_sync(audio: Path, hint: str | None) -> list[Word]: ] -async def transcribe(audio: Path, hint: str | None = None) -> list[Word]: +async def transcribe(audio: Path) -> list[Word]: """Mots prononcés et leurs instants (Whisper, exécuté hors de la boucle async).""" - return await asyncio.to_thread(_transcribe_sync, audio, hint) + return await asyncio.to_thread(_transcribe_sync, audio) diff --git a/ffmpeg-service/app/api.py b/ffmpeg-service/app/api.py index 1c57e5a..0780a70 100644 --- a/ffmpeg-service/app/api.py +++ b/ffmpeg-service/app/api.py @@ -418,7 +418,7 @@ async def _caption_words_without_voice(display: str, video: Path, info, plan: re sinon texte réparti sur la durée (hors effet de fin).""" if info.has_audio: try: - spoken = await align.transcribe(video, hint=display) + spoken = await align.transcribe(video) if len(spoken) >= max(2, len(display.split()) // 3): return align.align_words(display, spoken, info.duration) except Exception as error: # noqa: BLE001 — on retombe sur la répartition diff --git a/ffmpeg-service/app/config.py b/ffmpeg-service/app/config.py index 9add083..2bc74ea 100644 --- a/ffmpeg-service/app/config.py +++ b/ffmpeg-service/app/config.py @@ -14,6 +14,12 @@ TEMP_DIR = Path(os.environ.get("TEMP_DIR", "/tmp/ffmpeg_processing")) FILE_TTL_SECONDS = int(os.environ.get("FILE_TTL_SECONDS", "3600")) GEMINI_TTS_MODEL = os.environ.get("GEMINI_TTS_MODEL", "gemini-2.5-flash-preview-tts") +# Gemini tire sa lecture au hasard : une température basse et une graine fixe +# donnent une intonation stable d'une génération à l'autre. +GEMINI_TTS_TEMPERATURE = float(os.environ.get("GEMINI_TTS_TEMPERATURE", "0.3")) +GEMINI_TTS_SEED = int(os.environ.get("GEMINI_TTS_SEED", "20260928")) +# Générations tentées quand la lecture ne correspond pas au texte +GEMINI_TTS_ATTEMPTS = int(os.environ.get("GEMINI_TTS_ATTEMPTS", "3")) # Modèle Whisper utilisé pour caler les sous-titres sur la voix. WHISPER_MODEL = os.environ.get("WHISPER_MODEL", "base") diff --git a/ffmpeg-service/app/proc.py b/ffmpeg-service/app/proc.py index 185b97c..0f37302 100644 --- a/ffmpeg-service/app/proc.py +++ b/ffmpeg-service/app/proc.py @@ -20,8 +20,9 @@ class CommandError(RuntimeError): self.stderr = stderr -async def run(cmd: list[str], timeout: float = 900) -> str: - """Lance une commande et renvoie sa sortie standard.""" +async def run(cmd: list[str], timeout: float = 900, *, stderr_output: bool = False) -> str: + """Lance une commande et renvoie sa sortie standard (ou d'erreur, où FFmpeg + écrit les mesures de ses filtres d'analyse).""" log.debug("exec: %s", " ".join(cmd)) process = await asyncio.create_subprocess_exec( *cmd, stdout=asyncio.subprocess.PIPE, stderr=asyncio.subprocess.PIPE @@ -34,7 +35,7 @@ async def run(cmd: list[str], timeout: float = 900) -> str: raise CommandError(cmd, -1, f"délai de {timeout:.0f} s dépassé") from error if process.returncode != 0: raise CommandError(cmd, process.returncode, stderr.decode(errors="replace")) - return stdout.decode(errors="replace") + return (stderr if stderr_output else stdout).decode(errors="replace") @dataclass diff --git a/ffmpeg-service/app/quality.py b/ffmpeg-service/app/quality.py new file mode 100644 index 0000000..a382bca --- /dev/null +++ b/ffmpeg-service/app/quality.py @@ -0,0 +1,118 @@ +"""Contrôles de la voix générée : texte réellement lu et silences. + +Gemini lit parfois mal : mots sautés, fin tronquée, consigne de style lue à +voix haute. On compare donc ce que Whisper entend au texte attendu, et on +découpe la voix dans ses vrais silences (mesurés dans l'audio) plutôt qu'aux +bornes des mots estimées par Whisper, trop imprécises pour ne pas rogner une +syllabe. +""" + +import difflib +import re +from dataclasses import dataclass +from pathlib import Path + +from . import proc +from .align import Word, normalize + +# Seuils de silence : en dessous de -40 dB pendant au moins 120 ms +SILENCE_NOISE_DB = -40 +SILENCE_MIN_SECONDS = 0.12 + + +@dataclass +class ReadingCheck: + coverage: float # part des mots attendus effectivement entendus + extra: float # mots entendus en trop (consigne lue, ajouts), rapportés au texte + ending_ok: bool # la fin du texte est bien lue + + @property + def acceptable(self) -> bool: + return self.coverage >= 0.85 and self.extra <= 0.25 and self.ending_ok + + def describe(self) -> str: + parts = [f"{self.coverage:.0%} des mots lus"] + if not self.ending_ok: + parts.append("fin tronquée") + if self.extra > 0.25: + parts.append("mots ajoutés") + return ", ".join(parts) + + +def check_reading(expected_text: str, spoken: list[Word]) -> ReadingCheck: + """Compare le texte attendu à ce qui a été entendu (Whisper).""" + expected = [t for t in (normalize(w) for w in expected_text.split()) if t] + heard = [t for t in (normalize(w.text) for w in spoken) if t] + if not expected: + return ReadingCheck(1.0, 0.0, True) + if not heard: + return ReadingCheck(0.0, 0.0, False) + + matcher = difflib.SequenceMatcher(a=expected, b=heard, autojunk=False) + blocks = [b for b in matcher.get_matching_blocks() if b.size] + matched = sum(b.size for b in blocks) + # Fin lue : un des deux derniers mots attendus est reconnu + tail = {len(expected) - 1, len(expected) - 2} + ending_ok = any(b.a <= i < b.a + b.size for b in blocks for i in tail if i >= 0) + return ReadingCheck( + coverage=matched / len(expected), + extra=max(0, len(heard) - matched) / len(expected), + ending_ok=ending_ok, + ) + + +_SILENCE_START = re.compile(r"silence_start: (-?[\d.]+)") +_SILENCE_END = re.compile(r"silence_end: (-?[\d.]+)") + + +def parse_silences(ffmpeg_log: str, duration: float) -> list[tuple[float, float]]: + """Intervalles de silence tirés de la sortie du filtre silencedetect.""" + silences: list[tuple[float, float]] = [] + start: float | None = None + for line in ffmpeg_log.splitlines(): + if m := _SILENCE_START.search(line): + start = max(0.0, float(m.group(1))) + elif (m := _SILENCE_END.search(line)) and start is not None: + silences.append((start, float(m.group(1)))) + start = None + if start is not None: + silences.append((start, duration)) + return silences + + +async def detect_silences(path: Path, duration: float) -> list[tuple[float, float]]: + log = await proc.run( + [ + "ffmpeg", "-hide_banner", "-nostats", "-i", str(path), + "-af", f"silencedetect=noise={SILENCE_NOISE_DB}dB:d={SILENCE_MIN_SECONDS}", + "-f", "null", "-", + ], + timeout=120, + stderr_output=True, + ) # fmt: skip + return parse_silences(log, duration) + + +def speech_bounds( + silences: list[tuple[float, float]], duration: float, pad: float = 0.08 +) -> tuple[float, float]: + """Début et fin de la parole (silences d'ouverture et de fin retirés, avec une marge).""" + start, end = 0.0, duration + for s, e in silences: + if s <= 0.01: + start = max(start, e) + if e >= duration - 0.01: + end = min(end, s) + start, end = max(0.0, start - pad), min(duration, end + pad) + return (start, end) if end > start else (0.0, duration) + + +def cut_point(silences: list[tuple[float, float]], after: float, before: float) -> float: + """Instant de coupe entre deux phrases : au milieu du plus long silence + situé entre la fin de l'une et le début de l'autre (bornes Whisper élargies, + car imprécises). Sans silence mesuré, le milieu de l'écart.""" + low, high = after - 0.3, before + 0.3 + candidates = [(e - s, (s + e) / 2) for s, e in silences if e > low and s < high] + if candidates: + return max(candidates)[1] + return (after + before) / 2 diff --git a/ffmpeg-service/app/srt_voice.py b/ffmpeg-service/app/srt_voice.py index 1ad41ef..fa0c128 100644 --- a/ffmpeg-service/app/srt_voice.py +++ b/ffmpeg-service/app/srt_voice.py @@ -1,29 +1,41 @@ """Voix calée sur un fichier SRT. -Chaque sous-titre est lu séparément puis posé à son instant de début. Si la -lecture est plus longue que le sous-titre, elle est accélérée (sans changer -la hauteur de la voix) pour finir à temps : la voix respecte le minutage du -fichier, au lieu d'un texte lu d'une traite. +Tout le texte est lu d'une seule traite, puis la voix est découpée phrase par +phrase dans ses vrais silences et chaque morceau est posé à l'instant de son +sous-titre. Une lecture unique garde la même intonation, le même rythme et le +même niveau d'un bout à l'autre ; lire chaque sous-titre séparément donnait +une voix différente à chaque phrase, et le découpage aux bornes estimées par +Whisper rognait des syllabes. + +Si la lecture unique ne peut pas être répartie entre les sous-titres, on lit +chaque sous-titre séparément (même voix, même graine), en découpant là aussi +dans les silences mesurés. + +Quand un morceau est plus long que son sous-titre, il est accéléré (sans +changer la hauteur de la voix) pour finir à temps. """ import logging +import re from dataclasses import dataclass from pathlib import Path -from . import proc, tts -from .align import Word +from . import proc, quality, tts +from .align import Word, display_tokens from .text import clean_text log = logging.getLogger(__name__) # Débordement toléré sur le silence qui suit un sous-titre avant d'accélérer OVERFLOW_TOLERANCE = 0.25 -# Marges gardées autour des mots lors du retrait des silences de la voix -TRIM_LEAD = 0.05 -TRIM_TAIL = 0.12 +# Marges gardées autour de la parole quand on retire les silences +LEAD_PAD = 0.08 +TAIL_PAD = 0.15 # Au-delà, l'accélération s'entend nettement : l'utilisateur est prévenu AUDIBLE_SPEEDUP = 1.35 +_SENTENCE_END = re.compile(r"[.!?…]$") + @dataclass class Cue: @@ -32,6 +44,15 @@ class Cue: text: str +@dataclass +class Segment: + """Morceau de la voix lue, à poser à l'instant d'un sous-titre.""" + + source_start: float + source_end: float + words: list[Word] + + def cue_windows(cues: list[Cue]) -> list[float]: """Durée disponible pour lire chaque sous-titre sans empiéter sur le suivant.""" windows = [] @@ -59,13 +80,51 @@ def atempo_chain(factor: float) -> str: return ",".join(f"atempo={s:.4f}" for s in steps) -def spoken_span(words: list[Word], duration: float) -> tuple[float, float]: - """Partie parlée de la voix : les silences de début et de fin sont retirés.""" - if not words: - return 0.0, duration - start = max(0.0, words[0].start - TRIM_LEAD) - end = min(duration, words[-1].end + TRIM_TAIL) - return (start, end) if end > start else (0.0, duration) +def as_sentence(text: str) -> str: + """Chaque sous-titre finit par une ponctuation : la voix y marque une pause, + ce qui laisse un silence où couper.""" + text = clean_text(text) + return text if _SENTENCE_END.search(text) else f"{text}." + + +def split_by_cue(words: list[Word], cue_texts: list[str]) -> list[list[Word]] | None: + """Répartit les mots de la lecture complète entre les sous-titres + (les mots affichés suivent l'ordre du texte). None si les comptes divergent.""" + counts = [len(display_tokens(t)) for t in cue_texts] + if sum(counts) != len(words) or 0 in counts: + return None + groups, index = [], 0 + for count in counts: + groups.append(words[index : index + count]) + index += count + return groups + + +def segments_from_reading( + groups: list[list[Word]], silences: list[tuple[float, float]], duration: float +) -> list[Segment] | None: + """Découpe la lecture complète entre les phrases, au milieu des silences.""" + start, end = quality.speech_bounds(silences, duration, pad=LEAD_PAD) + cuts = [start] + for previous, following in zip(groups, groups[1:]): + cuts.append(quality.cut_point(silences, previous[-1].end, following[0].start)) + cuts.append(end) + if any(b <= a for a, b in zip(cuts, cuts[1:])): + return None + + segments = [] + for i, group in enumerate(groups): + seg_start, seg_end = cuts[i], cuts[i + 1] + # Retire le silence de part et d'autre de la coupe, en gardant une marge + for s, e in silences: + if s <= seg_start < e: + seg_start = max(seg_start, e - LEAD_PAD) + if s < seg_end <= e: + seg_end = min(seg_end, s + TAIL_PAD) + if seg_end <= seg_start: + seg_start, seg_end = cuts[i], cuts[i + 1] + segments.append(Segment(seg_start, seg_end, group)) + return segments async def synthesize_cues( @@ -82,75 +141,89 @@ async def synthesize_cues( if not cues: raise ValueError("Aucun sous-titre lisible dans le fichier SRT") - warnings: list[str] = [] - windows = cue_windows(cues) - segments: list[tuple[Path, float]] = [] + options = dict(engine=engine, voice=voice, style=style, gemini_api_key=gemini_api_key) + texts = [as_sentence(c.text) for c in cues] + + # 1. Lecture d'une seule traite, découpée dans les silences + script = " ".join(texts) + reading = await tts.synthesize(text=script, display_text=script, workdir=workdir, **options) + silences = await quality.detect_silences(reading.path, reading.duration) + groups = split_by_cue(reading.words, texts) + segments = segments_from_reading(groups, silences, reading.duration) if groups else None + + if segments: + sources = [(reading.path, seg) for seg in segments] + warnings = list(reading.warnings) + used_engine, used_voice = reading.engine, reading.voice + else: + # 2. Repli : chaque sous-titre lu séparément, avec la voix de la lecture complète + log.warning("Lecture SRT non répartie entre les sous-titres : lecture phrase par phrase") + sources, warnings = [], list(reading.warnings) + used_engine, used_voice = reading.engine, reading.voice + for index, text in enumerate(texts): + cue_dir = workdir / f"cue_{index:03d}" + cue_dir.mkdir(exist_ok=True) + track = await tts.synthesize( + text=text, + display_text=text, + workdir=cue_dir, + **{**options, "engine": used_engine, "voice": used_voice or voice}, + ) + warnings += [w for w in track.warnings if w not in warnings] + cue_silences = await quality.detect_silences(track.path, track.duration) + span = quality.speech_bounds(cue_silences, track.duration, pad=LEAD_PAD) + sources.append((track.path, Segment(span[0], span[1], track.words))) + + placed: list[tuple[Path, float]] = [] words: list[Word] = [] - used_engine, used_voice = engine or "gemini", voice or "" - - for index, (cue, window) in enumerate(zip(cues, windows)): - cue_dir = workdir / f"cue_{index:03d}" - cue_dir.mkdir(exist_ok=True) - text = clean_text(cue.text) - track = await tts.synthesize( - text=text, - display_text=text, - engine=used_engine, - voice=voice, - style=style, - gemini_api_key=gemini_api_key, - workdir=cue_dir, - ) - for warning in track.warnings: - if warning not in warnings: - warnings.append(warning) - # Après un repli sur Edge, on reste sur Edge : une seule voix du début à la fin - used_engine, used_voice = track.engine, track.voice - - span_start, span_end = spoken_span(track.words, track.duration) - factor = speed_factor(span_end - span_start, window) + for index, (cue, window, (source, segment)) in enumerate(zip(cues, cue_windows(cues), sources)): + factor = speed_factor(segment.source_end - segment.source_start, window) if factor > AUDIBLE_SPEEDUP: warnings.append( f"Sous-titre {index + 1} (« {cue.text[:40]} ») lu {factor:.1f}× plus vite " "pour tenir dans son minutage : raccourcissez-le ou allongez sa durée." ) + fitted = workdir / f"segment_{index:03d}.wav" + await proc.run(extract_command(source, segment, factor, fitted), timeout=120) + placed.append((fitted, cue.start)) + added_period = not _SENTENCE_END.search(clean_text(cue.text)) + for position, w in enumerate(segment.words): + text = w.text + if added_period and position == len(segment.words) - 1: + text = text.rstrip(".") # point ajouté pour la lecture, pas à afficher + start = cue.start + max(0.0, w.start - segment.source_start) / factor + end = cue.start + max(0.0, w.end - segment.source_start) / factor + words.append(Word(text or w.text, start, min(max(end, start + 0.05), cue.start + window))) - fitted = cue_dir / "fitted.wav" - trim = f"atrim=start={span_start:.3f}:end={span_end:.3f}" - await proc.run( - [ - "ffmpeg", - "-y", - "-hide_banner", - "-loglevel", - "error", - "-i", - str(track.path), - "-af", - f"{trim},asetpts=PTS-STARTPTS,{atempo_chain(factor)}", - "-ac", - "1", - "-ar", - "48000", - "-c:a", - "pcm_s16le", - str(fitted), - ], - timeout=120, - ) - segments.append((fitted, cue.start)) - for w in track.words: - start = cue.start + max(0.0, w.start - span_start) / factor - end = cue.start + max(0.0, w.end - span_start) / factor - words.append(Word(w.text, start, min(max(end, start + 0.05), cue.start + window))) - - output = workdir / "voice.wav" - await proc.run(mix_command(segments, output), timeout=300) + output = workdir / "voice_srt.wav" + await proc.run(mix_command(placed, output), timeout=300) duration = (await proc.probe(output)).duration - log.info("Voix SRT prête : %s/%s, %d sous-titres, %.1f s", used_engine, used_voice, len(cues), duration) + log.info( + "Voix SRT prête : %s/%s, %d sous-titres, %.1f s (%s)", + used_engine, + used_voice, + len(cues), + duration, + "lecture unique" if segments else "phrase par phrase", + ) return tts.VoiceTrack(output, duration, _monotonic(words), used_engine, used_voice, warnings) +def extract_command(source: Path, segment: Segment, factor: float, output: Path) -> list[str]: + """Extrait un morceau de voix, avec un fondu de 10 ms aux bords (pas de clic) + et l'accélération éventuelle.""" + length = segment.source_end - segment.source_start + fade_out = max(0.0, length - 0.01) + filters = ( + f"atrim=start={segment.source_start:.3f}:end={segment.source_end:.3f},asetpts=PTS-STARTPTS," + f"afade=t=in:d=0.01,afade=t=out:st={fade_out:.3f}:d=0.01,{atempo_chain(factor)}" + ) + return [ + "ffmpeg", "-y", "-hide_banner", "-loglevel", "error", "-i", str(source), + "-af", filters, "-ac", "1", "-ar", "48000", "-c:a", "pcm_s16le", str(output), + ] # fmt: skip + + def mix_command(segments: list[tuple[Path, float]], output: Path) -> list[str]: """Pose chaque segment de voix à son instant, sur une seule piste.""" command = ["ffmpeg", "-y", "-hide_banner", "-loglevel", "error"] @@ -162,18 +235,9 @@ def mix_command(segments: list[tuple[Path, float]], output: Path) -> list[str]: f"{labels}amix=inputs={len(segments)}:duration=longest:dropout_transition=0:normalize=0[voice]" ) return command + [ - "-filter_complex", - ";".join(graph), - "-map", - "[voice]", - "-ac", - "1", - "-ar", - "48000", - "-c:a", - "pcm_s16le", - str(output), - ] + "-filter_complex", ";".join(graph), "-map", "[voice]", + "-ac", "1", "-ar", "48000", "-c:a", "pcm_s16le", str(output), + ] # fmt: skip def _monotonic(words: list[Word]) -> list[Word]: diff --git a/ffmpeg-service/app/tts/__init__.py b/ffmpeg-service/app/tts/__init__.py index 0e8ce44..5b2e55b 100644 --- a/ffmpeg-service/app/tts/__init__.py +++ b/ffmpeg-service/app/tts/__init__.py @@ -4,7 +4,7 @@ import logging from dataclasses import dataclass, field from pathlib import Path -from .. import align, audio, proc +from .. import align, audio, config, proc, quality from ..align import Word from . import edge, gemini @@ -32,7 +32,7 @@ async def synthesize( workdir: Path, ) -> VoiceTrack: warnings: list[str] = [] - raw: Path | None = None + processed: Path | None = None spoken: list[Word] = [] used_engine, used_voice = "edge", "" @@ -40,24 +40,65 @@ async def synthesize( if not gemini_api_key: warnings.append("Clé Gemini absente : voix Edge utilisée à la place.") else: - try: - raw, used_voice = await gemini.synthesize(text, voice, style, gemini_api_key, workdir) + result = await _gemini_checked(text, voice, style, gemini_api_key, workdir, warnings) + if result: + processed, spoken, used_voice = result used_engine = "gemini" - except Exception as error: # noqa: BLE001 — repli sur Edge - log.warning("Gemini TTS en échec, repli sur Edge : %s", error) - warnings.append(f"Gemini indisponible ({error}) : voix Edge utilisée à la place.") - if raw is None: + if processed is None: raw, spoken, used_voice = await edge.synthesize(text, voice, style, workdir) + processed = workdir / "voice.wav" + await audio.process_voice(raw, processed) - processed = workdir / "voice.wav" - await audio.process_voice(raw, processed) duration = (await proc.probe(processed)).duration - if not spoken: - # Gemini ne donne pas le minutage : Whisper le retrouve dans l'audio - spoken = await align.transcribe(processed, hint=text) + spoken = await align.transcribe(processed) words = align.align_words(display_text, spoken, duration) log.info("Voix prête : %s/%s, %.1f s, %d mots", used_engine, used_voice, duration, len(words)) return VoiceTrack(processed, duration, words, used_engine, used_voice, warnings) + + +async def _gemini_checked( + text: str, voice: str | None, style: str | None, api_key: str, workdir: Path, warnings: list[str] +) -> tuple[Path, list[Word], str] | None: + """Voix Gemini dont la lecture a été vérifiée : Whisper réécoute chaque + génération ; mots sautés, fin tronquée ou consigne lue à voix haute + déclenchent une nouvelle génération. Renvoie None pour se replier sur Edge.""" + best: tuple[quality.ReadingCheck, Path, list[Word], str] | None = None + last_error: Exception | None = None + + for attempt in range(max(1, config.GEMINI_TTS_ATTEMPTS)): + try: + raw, gemini_voice = await gemini.synthesize(text, voice, style, api_key, workdir, attempt) + except Exception as error: # noqa: BLE001 — nouvelle tentative, puis repli sur Edge + log.warning("Gemini TTS en échec (tentative %d) : %s", attempt + 1, error) + last_error = error + continue + + processed = workdir / f"voice_gemini_{attempt}.wav" + await audio.process_voice(raw, processed) + spoken = await align.transcribe(processed) + check = quality.check_reading(text, spoken) + log.info("Lecture Gemini (tentative %d) : %s", attempt + 1, check.describe()) + + if best is None or _score(check) > _score(best[0]): + best = (check, processed, spoken, gemini_voice) + if check.acceptable: + break + + if best is None: + warnings.append(f"Gemini indisponible ({last_error}) : voix Edge utilisée à la place.") + return None + + check, processed, spoken, gemini_voice = best + if not check.acceptable: + if check.coverage < 0.7: + warnings.append(f"Gemini a mal lu le texte ({check.describe()}) : voix Edge utilisée à la place.") + return None + warnings.append(f"Lecture Gemini imparfaite ({check.describe()}) : écoutez le résultat.") + return processed, spoken, gemini_voice + + +def _score(check: quality.ReadingCheck) -> float: + return check.coverage - check.extra + (0.2 if check.ending_ok else 0.0) diff --git a/ffmpeg-service/app/tts/gemini.py b/ffmpeg-service/app/tts/gemini.py index a4c6dd2..21d0cc5 100644 --- a/ffmpeg-service/app/tts/gemini.py +++ b/ffmpeg-service/app/tts/gemini.py @@ -21,19 +21,29 @@ class GeminiError(RuntimeError): def build_prompt(text: str, style: str | None) -> str: - """Texte envoyé au modèle : la consigne de style précède le texte à lire.""" + """Texte envoyé au modèle : la consigne de style précède le texte à lire. + La consigne demande une lecture intégrale, sans ajout : le modèle sautait + parfois des mots ou tronquait la fin.""" instruction = style_instruction(style) - return f"{instruction} :\n{text}" if instruction else text + if not instruction: + return text + return f"{instruction}. Lis tout le texte, mot pour mot, sans rien ajouter :\n{text}" -async def _request(model: str, prompt: str, voice: str, api_key: str) -> dict: - payload = { - "contents": [{"parts": [{"text": prompt}]}], - "generationConfig": { - "responseModalities": ["AUDIO"], - "speechConfig": {"voiceConfig": {"prebuiltVoiceConfig": {"voiceName": voice}}}, - }, +# Paramètres de tirage refusés par un modèle : on ne les renvoie plus +_unsupported_sampling: set[str] = set() + + +async def _request(model: str, prompt: str, voice: str, api_key: str, seed: int) -> dict: + generation: dict = { + "responseModalities": ["AUDIO"], + "speechConfig": {"voiceConfig": {"prebuiltVoiceConfig": {"voiceName": voice}}}, } + if model not in _unsupported_sampling: + # Intonation stable d'une génération à l'autre + generation["temperature"] = config.GEMINI_TTS_TEMPERATURE + generation["seed"] = seed + payload = {"contents": [{"parts": [{"text": prompt}]}], "generationConfig": generation} async with httpx.AsyncClient(timeout=120) as client: # Clé en en-tête : dans l'URL, elle finissait dans les journaux response = await client.post( @@ -41,27 +51,37 @@ async def _request(model: str, prompt: str, voice: str, api_key: str) -> dict: json=payload, headers={"x-goog-api-key": api_key}, ) + rejected_sampling = response.status_code == 400 and any( + word in response.text.lower() for word in ("seed", "temperature", "unknown name", "invalid json") + ) + if rejected_sampling and "temperature" in generation: + # Modèle qui refuse température ou graine : même requête sans eux + log.warning("Gemini %s refuse les paramètres de tirage : %s", model, response.text[:200]) + _unsupported_sampling.add(model) + return await _request(model, prompt, voice, api_key, seed) if response.status_code != 200: raise GeminiError(f"Gemini {model} : HTTP {response.status_code} {response.text[:300]}") return response.json() async def synthesize( - text: str, voice: str | None, style: str | None, api_key: str, workdir: Path + text: str, voice: str | None, style: str | None, api_key: str, workdir: Path, attempt: int = 0 ) -> tuple[Path, str]: - """Génère la voix ; renvoie un WAV brut et le nom de la voix utilisée.""" + """Génère la voix ; renvoie un WAV brut et le nom de la voix utilisée. + Chaque nouvelle tentative change de graine pour obtenir une autre lecture.""" + seed = config.GEMINI_TTS_SEED + attempt gemini_voice = resolve_gemini_voice(voice).id prompt = build_prompt(text, style) model = config.GEMINI_TTS_MODEL log.info("Gemini TTS : modèle=%s voix=%s style=%s (%d car.)", model, gemini_voice, style, len(text)) try: - data = await _request(model, prompt, gemini_voice, api_key) + data = await _request(model, prompt, gemini_voice, api_key, seed) except GeminiError as error: if model == FALLBACK_MODEL or "HTTP 404" not in str(error): raise log.warning("Modèle %s indisponible, repli sur %s", model, FALLBACK_MODEL) - data = await _request(FALLBACK_MODEL, prompt, gemini_voice, api_key) + data = await _request(FALLBACK_MODEL, prompt, gemini_voice, api_key, seed) try: part = next(p for p in data["candidates"][0]["content"]["parts"] if "inlineData" in p)["inlineData"] @@ -70,14 +90,14 @@ async def synthesize( audio = base64.b64decode(part["data"]) mime = part.get("mimeType", "") - raw = workdir / "gemini_raw.wav" + raw = workdir / f"gemini_raw_{attempt}.wav" if "wav" in mime: raw.write_bytes(audio) else: # PCM brut 16 bits (« audio/L16;codec=pcm;rate=24000 ») rate_match = re.search(r"rate=(\d+)", mime) - pcm = workdir / "gemini.pcm" + pcm = workdir / f"gemini_{attempt}.pcm" pcm.write_bytes(audio) await proc.run( [ diff --git a/ffmpeg-service/tests/test_quality.py b/ffmpeg-service/tests/test_quality.py new file mode 100644 index 0000000..0ff7758 --- /dev/null +++ b/ffmpeg-service/tests/test_quality.py @@ -0,0 +1,41 @@ +from app.align import Word +from app.quality import check_reading, cut_point, parse_silences, speech_bounds + + +def _heard(text: str) -> list[Word]: + return [Word(t, i * 0.3, i * 0.3 + 0.25) for i, t in enumerate(text.split())] + + +def test_complete_reading_is_accepted(): + check = check_reading( + "Découvrez nos nouveautés en magasin", _heard("découvrez nos nouveautés en magasin") + ) + assert check.acceptable and check.coverage == 1.0 + + +def test_truncated_ending_is_rejected(): + check = check_reading("Découvrez nos nouveautés en magasin", _heard("découvrez nos nouveautés")) + assert not check.ending_ok and not check.acceptable + + +def test_instruction_read_aloud_is_rejected(): + heard = _heard("lis ce texte sur un ton dynamique découvrez nos nouveautés en magasin") + assert not check_reading("Découvrez nos nouveautés en magasin", heard).acceptable + + +def test_silences_are_parsed_from_ffmpeg_log(): + log = ( + "[silencedetect] silence_start: 0\n[silencedetect] silence_end: 0.21 | silence_duration: 0.21\n" + "[silencedetect] silence_start: 1.5\n" + ) + assert parse_silences(log, 2.0) == [(0.0, 0.21), (1.5, 2.0)] + + +def test_speech_bounds_drop_leading_and_trailing_silence(): + start, end = speech_bounds([(0.0, 0.3), (1.5, 2.0)], 2.0, pad=0.1) + assert round(start, 3) == 0.2 and round(end, 3) == 1.6 + + +def test_cut_point_prefers_the_longest_nearby_silence(): + assert cut_point([(1.0, 1.1), (1.2, 1.8)], after=1.0, before=1.9) == 1.5 + assert cut_point([], after=1.0, before=2.0) == 1.5 diff --git a/ffmpeg-service/tests/test_srt_voice.py b/ffmpeg-service/tests/test_srt_voice.py index 4607424..b0ff3d8 100644 --- a/ffmpeg-service/tests/test_srt_voice.py +++ b/ffmpeg-service/tests/test_srt_voice.py @@ -2,7 +2,16 @@ from pathlib import Path from app.align import Word, spread_words from app.render import RenderPlan -from app.srt_voice import Cue, atempo_chain, cue_windows, mix_command, speed_factor, spoken_span +from app.srt_voice import ( + Cue, + as_sentence, + atempo_chain, + cue_windows, + mix_command, + segments_from_reading, + speed_factor, + split_by_cue, +) def test_window_stops_at_next_cue_and_tolerates_a_short_overflow(): @@ -20,10 +29,27 @@ def test_atempo_is_split_into_supported_steps(): assert atempo_chain(3.0) == "atempo=2.0000,atempo=1.5000" -def test_spoken_span_trims_silences_around_words(): - words = [Word("Bonjour", 0.4, 0.9), Word("!", 0.9, 1.2)] - assert [round(t, 3) for t in spoken_span(words, 2.0)] == [0.35, 1.32] - assert spoken_span([], 2.0) == (0.0, 2.0) +def test_each_cue_becomes_a_sentence_for_the_reading(): + assert as_sentence("Venez vite") == "Venez vite." + assert as_sentence("Promo !") == "Promo !" + + +def test_words_of_the_single_reading_are_split_between_cues(): + words = [Word(t, i, i + 0.5) for i, t in enumerate(["Bonjour.", "Venez", "vite."])] + groups = split_by_cue(words, ["Bonjour.", "Venez vite."]) + assert [[w.text for w in g] for g in groups] == [["Bonjour."], ["Venez", "vite."]] + assert split_by_cue(words, ["Bonjour."]) is None + + +def test_reading_is_cut_inside_silences_not_on_word_bounds(): + # Whisper situe mal la fin de « Bonjour » (0,9 s) : la vraie parole finit à 1,05 s, + # puis silence jusqu'à 1,6 s. La coupe tombe dans ce silence, sans rogner le mot. + groups = [[Word("Bonjour.", 0.2, 0.9)], [Word("Venez", 1.7, 2.1), Word("vite.", 2.1, 2.6)]] + silences = [(0.0, 0.15), (1.05, 1.6), (2.7, 3.0)] + first, second = segments_from_reading(groups, silences, 3.0) + assert first.source_start < 0.15 and 1.05 < first.source_end <= 1.2 + assert 1.45 <= second.source_start < 1.6 and second.source_end > 2.7 + assert first.source_end < second.source_start def test_each_segment_is_placed_at_its_cue_start():