mirror of
https://github.com/R0m1k3/Socialflow.git
synced 2026-10-11 17:26:45 +02:00
SRT : la voix n'est plus jamais accélérée (atempo sans plafond, jusqu'à 2× et plus, rendait la voix incompréhensible). Un morceau plus long que son sous-titre déborde et décale les suivants ; les sous-titres suivent la voix, la vidéo s'allonge et un avertissement signale le décalage. Rythme : style par défaut « neutre », consignes Gemini sans « rythme entraînant », débit Edge des styles dynamique/promo ramené à +0 %/+3 %. Nouveau moteur « qwen » : service qwen-tts (Qwen3-TTS sur CPU, gratuit, français, ton piloté par consigne, clonage de voix depuis qwen-tts/voices). Même vérification Whisper que Gemini, généralisée ; repli qwen → gemini → edge, et une lecture sous 85 % du texte est écartée (mots sautés/faux). Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Upu97wMmsRkBoj6iVM4rH6
85 lines
3.0 KiB
Python
85 lines
3.0 KiB
Python
from pathlib import Path
|
|
|
|
import numpy as np
|
|
|
|
from app.align import Word, spread_words
|
|
from app.quality import Envelope
|
|
from app.render import RenderPlan
|
|
from app.srt_voice import (
|
|
Cue,
|
|
Segment,
|
|
as_sentence,
|
|
extract_command,
|
|
mix_command,
|
|
place_segments,
|
|
segments_from_reading,
|
|
split_by_cue,
|
|
)
|
|
|
|
|
|
def test_segments_start_at_their_cue_when_they_fit():
|
|
cues = [Cue(0.0, 2.0, "a"), Cue(2.5, 4.0, "b"), Cue(6.0, 7.0, "c")]
|
|
assert place_segments(cues, [2.0, 1.0, 1.0]) == [0.0, 2.5, 6.0]
|
|
|
|
|
|
def test_overflowing_segment_pushes_the_next_ones_instead_of_speeding_up():
|
|
cues = [Cue(0.0, 1.0, "a"), Cue(1.0, 2.0, "b"), Cue(5.0, 6.0, "c")]
|
|
starts = place_segments(cues, [2.0, 2.0, 1.0])
|
|
assert starts[0] == 0.0
|
|
assert abs(starts[1] - 2.05) < 1e-9 # après la fin du premier, à vitesse normale
|
|
assert starts[2] == 5.0 # rattrape le minutage dès que possible
|
|
|
|
|
|
def test_segments_are_never_sped_up():
|
|
cmd = extract_command(Path("in.wav"), Segment(1.0, 4.0, []), Path("out.wav"))
|
|
assert "atempo" not in cmd[cmd.index("-af") + 1]
|
|
|
|
|
|
def test_each_cue_becomes_a_sentence_for_the_reading():
|
|
assert as_sentence("Venez vite") == "Venez vite."
|
|
assert as_sentence("Promo !") == "Promo !"
|
|
|
|
|
|
def test_words_of_the_single_reading_are_split_between_cues():
|
|
words = [Word(t, i, i + 0.5) for i, t in enumerate(["Bonjour.", "Venez", "vite."])]
|
|
groups = split_by_cue(words, ["Bonjour.", "Venez vite."])
|
|
assert [[w.text for w in g] for g in groups] == [["Bonjour."], ["Venez", "vite."]]
|
|
assert split_by_cue(words, ["Bonjour."]) is None
|
|
|
|
|
|
def test_reading_is_cut_in_the_pause_without_clipping_sentence_endings():
|
|
# Whisper situe mal la fin de « Bonjour » (0,9 s) : la voix retombe
|
|
# doucement jusqu'à 1,05 s, puis vraie pause jusqu'à 1,6 s.
|
|
db = [-90.0] * 15 + [-20.0] * 75 + [-50.0] * 15 + [-90.0] * 55 + [-20.0] * 100 + [-90.0] * 40
|
|
env = Envelope(np.array(db))
|
|
groups = [[Word("Bonjour.", 0.2, 0.9)], [Word("Venez", 1.7, 2.1), Word("vite.", 2.1, 2.6)]]
|
|
first, second = segments_from_reading(groups, env)
|
|
assert first.source_end >= 1.05 + 0.25 # fin douce gardée, avec sa marge
|
|
assert first.source_end <= second.source_start
|
|
assert 1.5 <= second.source_start < 1.6
|
|
|
|
|
|
def test_each_segment_is_placed_at_its_cue_start():
|
|
cmd = mix_command([(Path("a.wav"), 0.0), (Path("b.wav"), 3.25)], Path("voice.wav"))
|
|
graph = cmd[cmd.index("-filter_complex") + 1]
|
|
assert "[1:a]adelay=3250:all=1[s1]" in graph
|
|
assert "amix=inputs=2" in graph and "normalize=0" in graph
|
|
|
|
|
|
def test_srt_voice_starts_without_delay():
|
|
plan = RenderPlan(
|
|
video=Path("in.mp4"),
|
|
video_duration=5.0,
|
|
output=Path("out.mp4"),
|
|
voice=Path("v.wav"),
|
|
voice_duration=8.0,
|
|
voice_delay=0.0,
|
|
)
|
|
assert plan.speech_end == 8.0
|
|
assert plan.total_duration == 8.8
|
|
|
|
|
|
def test_words_spread_over_their_cue():
|
|
words = spread_words("Bonjour à tous", 4.0, 5.0)
|
|
assert words[0].start == 4.0 and round(words[-1].end, 3) == 5.0
|