mirror of
https://github.com/R0m1k3/Socialflow.git
synced 2026-10-11 17:26:45 +02:00
feat(reels): import d'un fichier SRT lu par la voix à son minutage
- Nouveau Reel (bureau et mobile) : bouton « Importer un fichier SRT ». Tant qu'un SRT est chargé, le texte libre et l'assistant IA sont désactivés ; le choix de la voix (moteur, voix, ton) reste disponible. - Service Python : chaque sous-titre est lu séparément, ses silences retirés, puis posé à son instant de début. Une lecture trop longue est accélérée (atempo, sans changer la hauteur) pour tenir dans la durée du sous-titre, avec un avertissement au-delà de ×1,35. Après un repli sur Edge, tous les sous-titres gardent la même voix. - Sans voix, les mots de chaque sous-titre s'affichent sur son intervalle. - Aperçu : sous-titres calés sur le minutage du fichier. - Tests : lecture SRT (TS), fenêtres, accélération et mixage (Python). Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01WVqEw4Xycgf8BtwmSftE2M
This commit is contained in:
15 files changed
+635
-44
No files matched your search
@@ -90,6 +90,12 @@ def align_words(display_text: str, spoken: list[Word], total_duration: float | N
|
||||
return words
|
||||
|
||||
|
||||
def spread_words(text: str, start: float, end: float) -> list[Word]:
|
||||
"""Mots d'un texte répartis sur un intervalle (sous-titre SRT sans voix)."""
|
||||
tokens = display_tokens(text)
|
||||
return _spread(tokens, start, end) if tokens and end > start else []
|
||||
|
||||
|
||||
def _spread(tokens: list[str], start: float, end: float) -> list[Word]:
|
||||
"""Répartit des mots sur un intervalle au prorata de leur longueur."""
|
||||
weights = [max(1, len(normalize(t))) for t in tokens]
|
||||
|
||||
@@ -4,6 +4,7 @@ import asyncio
|
||||
import base64
|
||||
import contextlib
|
||||
import logging
|
||||
import shutil
|
||||
import time
|
||||
from pathlib import Path
|
||||
|
||||
@@ -12,7 +13,7 @@ from fastapi import Depends, FastAPI, Header, HTTPException
|
||||
from fastapi.responses import FileResponse
|
||||
from pydantic import BaseModel
|
||||
|
||||
from . import align, config, jobs, proc, render, subtitles, tts, voices
|
||||
from . import align, config, jobs, proc, render, srt_voice, subtitles, tts, voices
|
||||
from .audio import encode_preview
|
||||
from .text import clean_text
|
||||
|
||||
@@ -57,10 +58,18 @@ class TtsRequest(BaseModel):
|
||||
gemini_api_key: str | None = None
|
||||
|
||||
|
||||
class SrtCue(BaseModel):
|
||||
start: float
|
||||
end: float
|
||||
text: str
|
||||
|
||||
|
||||
class ReelRequest(BaseModel):
|
||||
video_base64: str | None = None
|
||||
video_url: str | None = None
|
||||
text: str | None = None
|
||||
# Sous-titres SRT : remplacent `text`, la voix lit chacun à son instant
|
||||
srt_cues: list[SrtCue] | None = None
|
||||
music_url: str | None = None
|
||||
watermark_url: str | None = None
|
||||
store_name: str | None = None
|
||||
@@ -182,6 +191,31 @@ async def _synthesize(text: str, display_source: str | None, request, workdir: P
|
||||
raise HTTPException(status_code=502, detail=f"La voix n'a pas pu être générée : {error}") from error
|
||||
|
||||
|
||||
def _srt_cues(request: ReelRequest) -> list[srt_voice.Cue]:
|
||||
"""Sous-titres lisibles du SRT, dans l'ordre (vide sans SRT)."""
|
||||
cues = [
|
||||
srt_voice.Cue(c.start, c.end, c.text)
|
||||
for c in request.srt_cues or []
|
||||
if c.end > c.start and clean_text(c.text)
|
||||
]
|
||||
return sorted(cues, key=lambda c: c.start)
|
||||
|
||||
|
||||
async def _synthesize_srt(cues: list[srt_voice.Cue], request, workdir: Path) -> tts.VoiceTrack:
|
||||
try:
|
||||
return await srt_voice.synthesize_cues(
|
||||
cues=cues,
|
||||
engine=request.tts_engine,
|
||||
voice=request.tts_voice,
|
||||
style=request.tts_style,
|
||||
gemini_api_key=request.gemini_api_key,
|
||||
workdir=workdir,
|
||||
)
|
||||
except Exception as error:
|
||||
log.exception("Voix SRT impossible à générer")
|
||||
raise HTTPException(status_code=502, detail=f"La voix n'a pas pu être générée : {error}") from error
|
||||
|
||||
|
||||
async def _download(url: str, target: Path, what: str, required: bool) -> bool:
|
||||
try:
|
||||
async with httpx.AsyncClient(timeout=httpx.Timeout(120, connect=15), follow_redirects=True) as client:
|
||||
@@ -240,8 +274,11 @@ async def _gather(request: ReelRequest, workdir: Path, clock: Stopwatch, *, fetc
|
||||
clock.lap("download")
|
||||
|
||||
track = None
|
||||
cues = _srt_cues(request)
|
||||
spoken_text = clean_text(request.text)
|
||||
if request.tts_enabled and spoken_text:
|
||||
if request.tts_enabled and cues:
|
||||
track = await _synthesize_srt(cues, request, workdir)
|
||||
elif request.tts_enabled and spoken_text:
|
||||
track = await _synthesize(spoken_text, request.text, request, workdir)
|
||||
clock.lap("tts")
|
||||
|
||||
@@ -254,6 +291,8 @@ async def _gather(request: ReelRequest, workdir: Path, clock: Stopwatch, *, fetc
|
||||
music_volume=request.music_volume,
|
||||
voice=track.path if track else None,
|
||||
voice_duration=track.duration if track else 0.0,
|
||||
# Avec un SRT, la voix est déjà posée aux instants du fichier
|
||||
voice_delay=0.0 if cues else config.VOICE_DELAY,
|
||||
watermark=watermark if has_watermark else None,
|
||||
outro_expected=not fetch_logo and request.has_logo,
|
||||
ending_effect=request.enable_ending_effect,
|
||||
@@ -273,6 +312,14 @@ async def _gather(request: ReelRequest, workdir: Path, clock: Stopwatch, *, fetc
|
||||
|
||||
async def _caption_words(request: ReelRequest, plan: render.RenderPlan, track, info) -> list[align.Word]:
|
||||
"""Mots à afficher, en secondes depuis le début de la vidéo."""
|
||||
cues = _srt_cues(request)
|
||||
if cues:
|
||||
if not request.draw_text:
|
||||
return []
|
||||
if track:
|
||||
return track.words
|
||||
# Sans voix : chaque sous-titre s'affiche sur son propre intervalle
|
||||
return [w for c in cues for w in align.spread_words(clean_text(c.text), c.start, c.end)]
|
||||
display = clean_text(request.text)
|
||||
if not request.draw_text or not display:
|
||||
return []
|
||||
@@ -284,7 +331,11 @@ async def _caption_words(request: ReelRequest, plan: render.RenderPlan, track, i
|
||||
def _keep_only(workdir: Path, keep: set[Path]) -> None:
|
||||
"""Seuls les fichiers à télécharger restent jusqu'à la récupération."""
|
||||
for entry in workdir.iterdir():
|
||||
if entry not in keep:
|
||||
if entry in keep:
|
||||
continue
|
||||
if entry.is_dir():
|
||||
shutil.rmtree(entry, ignore_errors=True) # voix SRT : un dossier par sous-titre
|
||||
else:
|
||||
entry.unlink(missing_ok=True)
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,183 @@
|
||||
"""Voix calée sur un fichier SRT.
|
||||
|
||||
Chaque sous-titre est lu séparément puis posé à son instant de début. Si la
|
||||
lecture est plus longue que le sous-titre, elle est accélérée (sans changer
|
||||
la hauteur de la voix) pour finir à temps : la voix respecte le minutage du
|
||||
fichier, au lieu d'un texte lu d'une traite.
|
||||
"""
|
||||
|
||||
import logging
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
|
||||
from . import proc, tts
|
||||
from .align import Word
|
||||
from .text import clean_text
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
# Débordement toléré sur le silence qui suit un sous-titre avant d'accélérer
|
||||
OVERFLOW_TOLERANCE = 0.25
|
||||
# Marges gardées autour des mots lors du retrait des silences de la voix
|
||||
TRIM_LEAD = 0.05
|
||||
TRIM_TAIL = 0.12
|
||||
# Au-delà, l'accélération s'entend nettement : l'utilisateur est prévenu
|
||||
AUDIBLE_SPEEDUP = 1.35
|
||||
|
||||
|
||||
@dataclass
|
||||
class Cue:
|
||||
start: float
|
||||
end: float
|
||||
text: str
|
||||
|
||||
|
||||
def cue_windows(cues: list[Cue]) -> list[float]:
|
||||
"""Durée disponible pour lire chaque sous-titre sans empiéter sur le suivant."""
|
||||
windows = []
|
||||
for i, cue in enumerate(cues):
|
||||
limit = cue.end + OVERFLOW_TOLERANCE
|
||||
if i + 1 < len(cues):
|
||||
limit = min(limit, max(cue.end, cues[i + 1].start))
|
||||
windows.append(max(0.1, limit - cue.start))
|
||||
return windows
|
||||
|
||||
|
||||
def speed_factor(duration: float, window: float) -> float:
|
||||
"""Accélération nécessaire pour tenir dans la fenêtre (1 = vitesse normale)."""
|
||||
return max(1.0, duration / window) if window > 0 else 1.0
|
||||
|
||||
|
||||
def atempo_chain(factor: float) -> str:
|
||||
"""Filtre atempo ; découpé en étapes ≤ 2 pour les anciennes versions de FFmpeg."""
|
||||
steps = []
|
||||
remaining = factor
|
||||
while remaining > 2.0:
|
||||
steps.append(2.0)
|
||||
remaining /= 2.0
|
||||
steps.append(remaining)
|
||||
return ",".join(f"atempo={s:.4f}" for s in steps)
|
||||
|
||||
|
||||
def spoken_span(words: list[Word], duration: float) -> tuple[float, float]:
|
||||
"""Partie parlée de la voix : les silences de début et de fin sont retirés."""
|
||||
if not words:
|
||||
return 0.0, duration
|
||||
start = max(0.0, words[0].start - TRIM_LEAD)
|
||||
end = min(duration, words[-1].end + TRIM_TAIL)
|
||||
return (start, end) if end > start else (0.0, duration)
|
||||
|
||||
|
||||
async def synthesize_cues(
|
||||
*,
|
||||
cues: list[Cue],
|
||||
engine: str | None,
|
||||
voice: str | None,
|
||||
style: str | None,
|
||||
gemini_api_key: str | None,
|
||||
workdir: Path,
|
||||
) -> tts.VoiceTrack:
|
||||
"""Voix complète : mots minutés en secondes depuis le début de la vidéo."""
|
||||
cues = sorted((c for c in cues if clean_text(c.text)), key=lambda c: c.start)
|
||||
if not cues:
|
||||
raise ValueError("Aucun sous-titre lisible dans le fichier SRT")
|
||||
|
||||
warnings: list[str] = []
|
||||
windows = cue_windows(cues)
|
||||
segments: list[tuple[Path, float]] = []
|
||||
words: list[Word] = []
|
||||
used_engine, used_voice = engine or "gemini", voice or ""
|
||||
|
||||
for index, (cue, window) in enumerate(zip(cues, windows)):
|
||||
cue_dir = workdir / f"cue_{index:03d}"
|
||||
cue_dir.mkdir(exist_ok=True)
|
||||
text = clean_text(cue.text)
|
||||
track = await tts.synthesize(
|
||||
text=text,
|
||||
display_text=text,
|
||||
engine=used_engine,
|
||||
voice=voice,
|
||||
style=style,
|
||||
gemini_api_key=gemini_api_key,
|
||||
workdir=cue_dir,
|
||||
)
|
||||
for warning in track.warnings:
|
||||
if warning not in warnings:
|
||||
warnings.append(warning)
|
||||
# Après un repli sur Edge, on reste sur Edge : une seule voix du début à la fin
|
||||
used_engine, used_voice = track.engine, track.voice
|
||||
|
||||
span_start, span_end = spoken_span(track.words, track.duration)
|
||||
factor = speed_factor(span_end - span_start, window)
|
||||
if factor > AUDIBLE_SPEEDUP:
|
||||
warnings.append(
|
||||
f"Sous-titre {index + 1} (« {cue.text[:40]} ») lu {factor:.1f}× plus vite "
|
||||
"pour tenir dans son minutage : raccourcissez-le ou allongez sa durée."
|
||||
)
|
||||
|
||||
fitted = cue_dir / "fitted.wav"
|
||||
trim = f"atrim=start={span_start:.3f}:end={span_end:.3f}"
|
||||
await proc.run(
|
||||
[
|
||||
"ffmpeg",
|
||||
"-y",
|
||||
"-hide_banner",
|
||||
"-loglevel",
|
||||
"error",
|
||||
"-i",
|
||||
str(track.path),
|
||||
"-af",
|
||||
f"{trim},asetpts=PTS-STARTPTS,{atempo_chain(factor)}",
|
||||
"-ac",
|
||||
"1",
|
||||
"-ar",
|
||||
"48000",
|
||||
"-c:a",
|
||||
"pcm_s16le",
|
||||
str(fitted),
|
||||
],
|
||||
timeout=120,
|
||||
)
|
||||
segments.append((fitted, cue.start))
|
||||
for w in track.words:
|
||||
start = cue.start + max(0.0, w.start - span_start) / factor
|
||||
end = cue.start + max(0.0, w.end - span_start) / factor
|
||||
words.append(Word(w.text, start, min(max(end, start + 0.05), cue.start + window)))
|
||||
|
||||
output = workdir / "voice.wav"
|
||||
await proc.run(mix_command(segments, output), timeout=300)
|
||||
duration = (await proc.probe(output)).duration
|
||||
log.info("Voix SRT prête : %s/%s, %d sous-titres, %.1f s", used_engine, used_voice, len(cues), duration)
|
||||
return tts.VoiceTrack(output, duration, _monotonic(words), used_engine, used_voice, warnings)
|
||||
|
||||
|
||||
def mix_command(segments: list[tuple[Path, float]], output: Path) -> list[str]:
|
||||
"""Pose chaque segment de voix à son instant, sur une seule piste."""
|
||||
command = ["ffmpeg", "-y", "-hide_banner", "-loglevel", "error"]
|
||||
for path, _ in segments:
|
||||
command += ["-i", str(path)]
|
||||
graph = [f"[{i}:a]adelay={int(round(start * 1000))}:all=1[s{i}]" for i, (_, start) in enumerate(segments)]
|
||||
labels = "".join(f"[s{i}]" for i in range(len(segments)))
|
||||
graph.append(
|
||||
f"{labels}amix=inputs={len(segments)}:duration=longest:dropout_transition=0:normalize=0[voice]"
|
||||
)
|
||||
return command + [
|
||||
"-filter_complex",
|
||||
";".join(graph),
|
||||
"-map",
|
||||
"[voice]",
|
||||
"-ac",
|
||||
"1",
|
||||
"-ar",
|
||||
"48000",
|
||||
"-c:a",
|
||||
"pcm_s16le",
|
||||
str(output),
|
||||
]
|
||||
|
||||
|
||||
def _monotonic(words: list[Word]) -> list[Word]:
|
||||
for prev, cur in zip(words, words[1:]):
|
||||
cur.start = max(cur.start, prev.start + 0.01)
|
||||
cur.end = max(cur.end, cur.start + 0.05)
|
||||
return words
|
||||
@@ -0,0 +1,51 @@
|
||||
from pathlib import Path
|
||||
|
||||
from app.align import Word, spread_words
|
||||
from app.render import RenderPlan
|
||||
from app.srt_voice import Cue, atempo_chain, cue_windows, mix_command, speed_factor, spoken_span
|
||||
|
||||
|
||||
def test_window_stops_at_next_cue_and_tolerates_a_short_overflow():
|
||||
cues = [Cue(0.0, 2.0, "a"), Cue(2.1, 4.0, "b"), Cue(6.0, 7.0, "c")]
|
||||
assert cue_windows(cues) == [2.1, 2.15, 1.25]
|
||||
|
||||
|
||||
def test_voice_is_sped_up_only_when_too_long():
|
||||
assert speed_factor(1.5, 2.0) == 1.0
|
||||
assert speed_factor(3.0, 2.0) == 1.5
|
||||
|
||||
|
||||
def test_atempo_is_split_into_supported_steps():
|
||||
assert atempo_chain(1.25) == "atempo=1.2500"
|
||||
assert atempo_chain(3.0) == "atempo=2.0000,atempo=1.5000"
|
||||
|
||||
|
||||
def test_spoken_span_trims_silences_around_words():
|
||||
words = [Word("Bonjour", 0.4, 0.9), Word("!", 0.9, 1.2)]
|
||||
assert [round(t, 3) for t in spoken_span(words, 2.0)] == [0.35, 1.32]
|
||||
assert spoken_span([], 2.0) == (0.0, 2.0)
|
||||
|
||||
|
||||
def test_each_segment_is_placed_at_its_cue_start():
|
||||
cmd = mix_command([(Path("a.wav"), 0.0), (Path("b.wav"), 3.25)], Path("voice.wav"))
|
||||
graph = cmd[cmd.index("-filter_complex") + 1]
|
||||
assert "[1:a]adelay=3250:all=1[s1]" in graph
|
||||
assert "amix=inputs=2" in graph and "normalize=0" in graph
|
||||
|
||||
|
||||
def test_srt_voice_starts_without_delay():
|
||||
plan = RenderPlan(
|
||||
video=Path("in.mp4"),
|
||||
video_duration=5.0,
|
||||
output=Path("out.mp4"),
|
||||
voice=Path("v.wav"),
|
||||
voice_duration=8.0,
|
||||
voice_delay=0.0,
|
||||
)
|
||||
assert plan.speech_end == 8.0
|
||||
assert plan.total_duration == 8.8
|
||||
|
||||
|
||||
def test_words_spread_over_their_cue():
|
||||
words = spread_words("Bonjour à tous", 4.0, 5.0)
|
||||
assert words[0].start == 4.0 and round(words[-1].end, 3) == 5.0
|
||||
Reference in new issue
Block a user