Files
Claude 44daa88a49 Voix : plus d'accélération sur les SRT, moteur local Qwen3-TTS
SRT : la voix n'est plus jamais accélérée (atempo sans plafond, jusqu'à 2×
et plus, rendait la voix incompréhensible). Un morceau plus long que son
sous-titre déborde et décale les suivants ; les sous-titres suivent la voix,
la vidéo s'allonge et un avertissement signale le décalage.

Rythme : style par défaut « neutre », consignes Gemini sans « rythme
entraînant », débit Edge des styles dynamique/promo ramené à +0 %/+3 %.

Nouveau moteur « qwen » : service qwen-tts (Qwen3-TTS sur CPU, gratuit,
français, ton piloté par consigne, clonage de voix depuis qwen-tts/voices).
Même vérification Whisper que Gemini, généralisée ; repli qwen → gemini →
edge, et une lecture sous 85 % du texte est écartée (mots sautés/faux).

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01Upu97wMmsRkBoj6iVM4rH6
2026-09-29 09:39:47 +00:00

85 lines
3.0 KiB
Python

from pathlib import Path
import numpy as np
from app.align import Word, spread_words
from app.quality import Envelope
from app.render import RenderPlan
from app.srt_voice import (
Cue,
Segment,
as_sentence,
extract_command,
mix_command,
place_segments,
segments_from_reading,
split_by_cue,
)
def test_segments_start_at_their_cue_when_they_fit():
cues = [Cue(0.0, 2.0, "a"), Cue(2.5, 4.0, "b"), Cue(6.0, 7.0, "c")]
assert place_segments(cues, [2.0, 1.0, 1.0]) == [0.0, 2.5, 6.0]
def test_overflowing_segment_pushes_the_next_ones_instead_of_speeding_up():
cues = [Cue(0.0, 1.0, "a"), Cue(1.0, 2.0, "b"), Cue(5.0, 6.0, "c")]
starts = place_segments(cues, [2.0, 2.0, 1.0])
assert starts[0] == 0.0
assert abs(starts[1] - 2.05) < 1e-9 # après la fin du premier, à vitesse normale
assert starts[2] == 5.0 # rattrape le minutage dès que possible
def test_segments_are_never_sped_up():
cmd = extract_command(Path("in.wav"), Segment(1.0, 4.0, []), Path("out.wav"))
assert "atempo" not in cmd[cmd.index("-af") + 1]
def test_each_cue_becomes_a_sentence_for_the_reading():
assert as_sentence("Venez vite") == "Venez vite."
assert as_sentence("Promo !") == "Promo !"
def test_words_of_the_single_reading_are_split_between_cues():
words = [Word(t, i, i + 0.5) for i, t in enumerate(["Bonjour.", "Venez", "vite."])]
groups = split_by_cue(words, ["Bonjour.", "Venez vite."])
assert [[w.text for w in g] for g in groups] == [["Bonjour."], ["Venez", "vite."]]
assert split_by_cue(words, ["Bonjour."]) is None
def test_reading_is_cut_in_the_pause_without_clipping_sentence_endings():
# Whisper situe mal la fin de « Bonjour » (0,9 s) : la voix retombe
# doucement jusqu'à 1,05 s, puis vraie pause jusqu'à 1,6 s.
db = [-90.0] * 15 + [-20.0] * 75 + [-50.0] * 15 + [-90.0] * 55 + [-20.0] * 100 + [-90.0] * 40
env = Envelope(np.array(db))
groups = [[Word("Bonjour.", 0.2, 0.9)], [Word("Venez", 1.7, 2.1), Word("vite.", 2.1, 2.6)]]
first, second = segments_from_reading(groups, env)
assert first.source_end >= 1.05 + 0.25 # fin douce gardée, avec sa marge
assert first.source_end <= second.source_start
assert 1.5 <= second.source_start < 1.6
def test_each_segment_is_placed_at_its_cue_start():
cmd = mix_command([(Path("a.wav"), 0.0), (Path("b.wav"), 3.25)], Path("voice.wav"))
graph = cmd[cmd.index("-filter_complex") + 1]
assert "[1:a]adelay=3250:all=1[s1]" in graph
assert "amix=inputs=2" in graph and "normalize=0" in graph
def test_srt_voice_starts_without_delay():
plan = RenderPlan(
video=Path("in.mp4"),
video_duration=5.0,
output=Path("out.mp4"),
voice=Path("v.wav"),
voice_duration=8.0,
voice_delay=0.0,
)
assert plan.speech_end == 8.0
assert plan.total_duration == 8.8
def test_words_spread_over_their_cue():
words = spread_words("Bonjour à tous", 4.0, 5.0)
assert words[0].start == 4.0 and round(words[-1].end, 3) == 5.0