feat(reels): import d'un fichier SRT lu par la voix à son minutage

- Nouveau Reel (bureau et mobile) : bouton « Importer un fichier SRT ».
  Tant qu'un SRT est chargé, le texte libre et l'assistant IA sont
  désactivés ; le choix de la voix (moteur, voix, ton) reste disponible.
- Service Python : chaque sous-titre est lu séparément, ses silences
  retirés, puis posé à son instant de début. Une lecture trop longue est
  accélérée (atempo, sans changer la hauteur) pour tenir dans la durée du
  sous-titre, avec un avertissement au-delà de ×1,35. Après un repli sur
  Edge, tous les sous-titres gardent la même voix.
- Sans voix, les mots de chaque sous-titre s'affichent sur son intervalle.
- Aperçu : sous-titres calés sur le minutage du fichier.
- Tests : lecture SRT (TS), fenêtres, accélération et mixage (Python).

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01WVqEw4Xycgf8BtwmSftE2M
This commit is contained in:
Claude committed 2026-09-27 08:19:44 +00:00
1 parent 9b66a6dc5a
commit f5e2461da1
15 files changed
+635 -44

No files matched your search

+18 -6
View File
@@ -14,6 +14,7 @@ import {
type CaptionStyle,
type TimedWord,
} from "@shared/captions";
import { srtEnd, srtWords, type SrtCue } from "@shared/srt";
import { ReelVideo, type ReelVideoProps } from "@/remotion/ReelVideo";
import { ImageComposition, type ImageCompositionProps } from "@/remotion/ImageComposition";
import type { VoicePreviewResult } from "./voice-picker";
@@ -33,6 +34,8 @@ interface CommonProps {
logoUrl?: string | null;
storeName?: string;
endingEffect: boolean;
/** Sous-titres SRT : remplacent `text`, minutés par le fichier. */
srtCues?: SrtCue[] | null;
}
type ReelPreviewProps =
@@ -74,9 +77,12 @@ export function ReelPreview(props: ReelPreviewProps) {
const { text, showCaptions, captionStyle, ttsEnabled, voice, musicUrl, musicVolume, endingEffect } = props;
const logoUrl = props.logoUrl ?? undefined;
const storeName = endingEffect ? props.storeName || undefined : undefined;
const srtCues = props.kind === "video" && props.srtCues?.length ? props.srtCues : null;
const cleanText = cleanCaptionText(text);
const voiceDuration = ttsEnabled && cleanText ? voice?.duration ?? estimatedVoiceDuration(text) : 0;
const estimated = ttsEnabled && Boolean(cleanText) && !voice;
const voiceDuration = srtCues
? ttsEnabled ? srtEnd(srtCues) : 0
: ttsEnabled && cleanText ? voice?.duration ?? estimatedVoiceDuration(text) : 0;
const estimated = !srtCues && ttsEnabled && Boolean(cleanText) && !voice;
const composition = useMemo(() => {
if (props.kind === "images") {
@@ -112,9 +118,13 @@ export function ReelPreview(props: ReelPreviewProps) {
voiceDuration: voiceDuration || undefined,
hasOutro: Boolean(logoUrl),
endingEffect,
voiceDelay: srtCues ? 0 : VOICE_DELAY,
});
let words: TimedWord[] = [];
if (showCaptions && cleanText) {
if (srtCues) {
// La voix de l'aperçu n'est pas générée sous-titre par sous-titre : minutage du fichier
if (showCaptions) words = srtWords(srtCues);
} else if (showCaptions && cleanText) {
if (voice) words = offsetWords(voice.words, VOICE_DELAY);
else if (ttsEnabled) words = spreadWords(text, VOICE_DELAY, VOICE_DELAY + voiceDuration);
else words = spreadWords(text, 0.5, (timing.logoStart ?? timing.total) - 0.5);
@@ -129,7 +139,7 @@ export function ReelPreview(props: ReelPreviewProps) {
storeName,
logoStart: timing.logoStart,
fadeStart: timing.fadeStart,
voiceUrl: voice?.audioUrl,
voiceUrl: srtCues ? undefined : voice?.audioUrl,
voiceDelay: VOICE_DELAY,
musicUrl,
musicVolume,
@@ -139,7 +149,7 @@ export function ReelPreview(props: ReelPreviewProps) {
}, [
props.kind,
props.kind === "video" ? props.videoUrl : props.images.join("|"),
videoDuration, text, cleanText, showCaptions, captionStyle, ttsEnabled, voice, voiceDuration,
videoDuration, text, cleanText, srtCues, showCaptions, captionStyle, ttsEnabled, voice, voiceDuration,
musicUrl, musicVolume, logoUrl, storeName, endingEffect,
]);
@@ -171,7 +181,9 @@ export function ReelPreview(props: ReelPreviewProps) {
</div>
<p className="text-xs text-muted-foreground flex items-start gap-1.5">
<Info className="w-3.5 h-3.5 mt-0.5 shrink-0" />
{estimated
{srtCues
? `Minutage du fichier SRT (la voix sera posée sur chaque sous-titre au rendu) · ${composition.total.toFixed(1)} s`
: estimated
? "Minutage estimé : cliquez sur « Tester la voix » pour caler les sous-titres sur la vraie voix."
: `Aperçu fidèle au rendu final · ${composition.total.toFixed(1)} s`}
</p>
@@ -0,0 +1,89 @@
import { useRef } from "react";
import { FileText, Upload, X } from "lucide-react";
import { Button } from "@/components/ui/button";
import { useToast } from "@/hooks/use-toast";
import { formatSrtTime, parseSrt, srtEnd, type SrtCue } from "@shared/srt";
const MAX_SRT_BYTES = 512 * 1024;
export interface SrtFile {
name: string;
cues: SrtCue[];
}
interface SrtUploadProps {
value: SrtFile | null;
onChange: (value: SrtFile | null) => void;
}
/**
* Import d'un fichier de sous-titres SRT. Tant qu'un fichier est chargé, il
* remplace le texte libre : la voix lit chaque sous-titre à son instant.
*/
export function SrtUpload({ value, onChange }: SrtUploadProps) {
const { toast } = useToast();
const inputRef = useRef<HTMLInputElement>(null);
const handleFile = async (file: File | undefined) => {
if (!file) return;
try {
if (file.size > MAX_SRT_BYTES) throw new Error("Fichier SRT trop volumineux (512 Ko au plus)");
const cues = parseSrt(await file.text());
onChange({ name: file.name, cues });
toast({ title: "Sous-titres importés", description: `${cues.length} sous-titre(s) · ${formatSrtTime(srtEnd(cues))}` });
} catch (error) {
toast({
title: "Fichier SRT invalide",
description: error instanceof Error ? error.message : "Lecture impossible",
variant: "destructive",
});
} finally {
if (inputRef.current) inputRef.current.value = "";
}
};
return (
<div className="space-y-2">
<input
ref={inputRef}
type="file"
accept=".srt,application/x-subrip,text/plain"
className="hidden"
onChange={(e) => handleFile(e.target.files?.[0])}
/>
{value ? (
<div className="rounded-lg border border-primary/30 bg-primary/5 p-3 space-y-2">
<div className="flex items-center justify-between gap-2">
<div className="flex items-center gap-2 min-w-0 text-sm font-medium">
<FileText className="w-4 h-4 shrink-0 text-primary" />
<span className="truncate">{value.name}</span>
</div>
<Button type="button" variant="ghost" size="sm" onClick={() => onChange(null)}>
<X className="w-4 h-4 mr-1" />
Retirer
</Button>
</div>
<p className="text-xs text-muted-foreground">
{value.cues.length} sous-titre(s) · jusqu'à {formatSrtTime(srtEnd(value.cues))}. La voix lira chaque
sous-titre à son instant et accélérera si besoin pour tenir dans sa durée.
</p>
<ul className="max-h-40 overflow-y-auto space-y-1 text-xs">
{value.cues.map((cue, i) => (
<li key={i} className="flex gap-2">
<span className="shrink-0 font-mono text-muted-foreground">
{formatSrtTime(cue.start)}–{formatSrtTime(cue.end)}
</span>
<span>{cue.text}</span>
</li>
))}
</ul>
</div>
) : (
<Button type="button" variant="outline" className="w-full" onClick={() => inputRef.current?.click()}>
<Upload className="w-4 h-4 mr-2" />
Importer un fichier SRT
</Button>
)}
</div>
);
}
+30 -15
View File
@@ -19,6 +19,8 @@ import { useToast } from "@/hooks/use-toast";
import { VoicePicker, isVoicePreviewCurrent, type VoicePreviewResult, type VoiceSettings } from "@/components/reels/voice-picker";
import { CaptionStylePicker } from "@/components/reels/caption-style-picker";
import { ReelPreview } from "@/components/reels/reel-preview";
import { SrtUpload, type SrtFile } from "@/components/reels/srt-upload";
import { srtText } from "@shared/srt";
import { DEFAULT_CAPTION_STYLE, type CaptionStyle } from "@shared/captions";
import { DEFAULT_TTS_STYLE, DEFAULT_VOICE } from "@shared/voices";
import { apiRequest, queryClient, handleUnauthorized } from "@/lib/queryClient";
@@ -56,6 +58,8 @@ export default function MobileNewReel() {
const [selectedVideo, setSelectedVideo] = useState<Media | null>(null);
const [selectedTrack, setSelectedTrack] = useState<MusicTrack | null>(null);
const [overlayText, setOverlayText] = useState('');
// Fichier SRT : remplace le texte libre, la voix suit son minutage
const [srtFile, setSrtFile] = useState<SrtFile | null>(null);
const [productInfo, setProductInfo] = useState('');
const [generatedVariants, setGeneratedVariants] = useState<any[]>([]);
const [selectedPages, setSelectedPages] = useState<string[]>([]);
@@ -92,7 +96,7 @@ export default function MobileNewReel() {
// Auto-calculate TTS sync
useEffect(() => {
if (!ttsEnabled || !overlayText.trim()) {
if (!ttsEnabled || !overlayText.trim() || srtFile) {
setSyncInfo(null);
return;
}
@@ -107,7 +111,7 @@ export default function MobileNewReel() {
.catch(() => setSyncInfo(null));
}, 800);
return () => clearTimeout(timer);
}, [overlayText, ttsEnabled, ttsEngine, ttsVoice]);
}, [overlayText, ttsEnabled, ttsEngine, ttsVoice, srtFile]);
const { data: pages = [] } = useQuery<SocialPage[]>({
@@ -262,8 +266,9 @@ export default function MobileNewReel() {
createReelMutation.mutate({
videoMediaId: selectedVideoId,
musicTrackId: selectedTrack?.id,
overlayText: overlayText,
description: overlayText,
overlayText: srtFile ? undefined : overlayText,
srtCues: srtFile?.cues,
description: srtFile ? srtText(srtFile.cues) : overlayText,
pageIds: selectedPages,
scheduledFor: scheduledDate?.toISOString(),
musicVolume: musicVolume[0] / 100,
@@ -429,6 +434,7 @@ export default function MobileNewReel() {
{/* TEXT STEP */}
{currentStep === 'text' && (
<div className="space-y-6">
{!srtFile && (
<Card>
<CardHeader className="pb-3">
<CardTitle className="text-base flex items-center"><Sparkles className="w-4 h-4 mr-2 text-primary" /> Assistant IA</CardTitle>
@@ -446,8 +452,9 @@ export default function MobileNewReel() {
</Button>
</CardContent>
</Card>
)}
{generatedVariants.length > 0 && (
{!srtFile && generatedVariants.length > 0 && (
<div className="space-y-3">
{generatedVariants.map((v, i) => (
<div key={i} className="bg-card p-3 rounded-lg border text-sm" onClick={() => { setOverlayText(v.text); toast({ title: "Texte appliqué" }); }}>
@@ -459,13 +466,20 @@ export default function MobileNewReel() {
<div className="space-y-2">
<label className="text-sm font-medium">Texte Overlay</label>
<Textarea
value={overlayText}
onChange={(e) => setOverlayText(e.target.value)}
placeholder="Texte sur la vidéo..."
className="text-lg"
rows={3}
/>
<SrtUpload value={srtFile} onChange={setSrtFile} />
{srtFile ? (
<p className="text-xs text-muted-foreground">
Texte libre désactivé : le fichier SRT est utilisé. Retirez-le pour écrire un texte.
</p>
) : (
<Textarea
value={overlayText}
onChange={(e) => setOverlayText(e.target.value)}
placeholder="Texte sur la vidéo..."
className="text-lg"
rows={3}
/>
)}
<div className="space-y-1.5 mt-3">
<Label className="text-xs font-medium">Style des sous-titres</Label>
<CaptionStylePicker value={captionStyle} onChange={setCaptionStyle} compact />
@@ -503,13 +517,13 @@ export default function MobileNewReel() {
<VoicePicker
value={voiceSettings}
onChange={setVoiceSettings}
sampleText={overlayText}
sampleText={srtFile ? srtFile.cues[0]?.text : overlayText}
onPreview={setVoicePreview}
compact
/>
</div>
{syncInfo && (
{syncInfo && !srtFile && (
<div className={`p-2 rounded-lg border mt-2 ${syncInfo.isHealthy ? 'bg-green-500/10 border-green-500/30' : 'bg-yellow-500/10 border-yellow-500/30'}`}>
<div className="flex items-center justify-between text-xs">
<span className="font-medium">Sync</span>
@@ -549,7 +563,8 @@ export default function MobileNewReel() {
<ReelPreview
kind="video"
videoUrl={selectedVideo.originalUrl}
text={overlayText}
text={srtFile ? srtText(srtFile.cues) : overlayText}
srtCues={srtFile?.cues}
showCaptions
captionStyle={captionStyle}
ttsEnabled={ttsEnabled}
+38 -18
View File
@@ -28,6 +28,8 @@ import { useToast } from "@/hooks/use-toast";
import { VoicePicker, isVoicePreviewCurrent, type VoicePreviewResult, type VoiceSettings } from "@/components/reels/voice-picker";
import { CaptionStylePicker } from "@/components/reels/caption-style-picker";
import { ReelPreview } from "@/components/reels/reel-preview";
import { SrtUpload, type SrtFile } from "@/components/reels/srt-upload";
import { srtText } from "@shared/srt";
import { DEFAULT_CAPTION_STYLE, type CaptionStyle } from "@shared/captions";
import { DEFAULT_TTS_STYLE, DEFAULT_VOICE } from "@shared/voices";
import { apiRequest, queryClient, handleUnauthorized, getErrorMessage } from "@/lib/queryClient";
@@ -65,6 +67,8 @@ export default function NewReel() {
const [selectedVideo, setSelectedVideo] = useState<Media | null>(null);
const [selectedTrack, setSelectedTrack] = useState<MusicTrack | null>(null);
const [overlayText, setOverlayText] = useState('');
// Fichier SRT : remplace le texte libre, la voix suit son minutage
const [srtFile, setSrtFile] = useState<SrtFile | null>(null);
const [productInfo, setProductInfo] = useState('');
const [generatedVariants, setGeneratedVariants] = useState<any[]>([]);
const [selectedPages, setSelectedPages] = useState<string[]>([]);
@@ -108,7 +112,7 @@ export default function NewReel() {
// Auto-calculate TTS sync when text changes
useEffect(() => {
if (!ttsEnabled || !overlayText.trim()) {
if (!ttsEnabled || !overlayText.trim() || srtFile) {
setSyncInfo(null);
return;
}
@@ -123,7 +127,7 @@ export default function NewReel() {
.catch(() => setSyncInfo(null));
}, 800);
return () => clearTimeout(timer);
}, [overlayText, ttsEnabled, ttsEngine, ttsVoice]);
}, [overlayText, ttsEnabled, ttsEngine, ttsVoice, srtFile]);
// État audio preview
@@ -405,8 +409,9 @@ export default function NewReel() {
createReelMutation.mutate({
videoMediaId: selectedVideoId,
musicTrackId: selectedTrack?.id,
overlayText: overlayText,
description: overlayText,
overlayText: srtFile ? undefined : overlayText,
srtCues: srtFile?.cues,
description: srtFile ? srtText(srtFile.cues) : overlayText,
pageIds: selectedPages,
scheduledFor: scheduledDate?.toISOString(),
musicVolume: musicVolume[0] / 100,
@@ -702,6 +707,7 @@ export default function NewReel() {
{/* ÉTAPE 3: Texte */}
{currentStep === 'text' && (
<>
{!srtFile && (
<Card className="rounded-2xl border-border/50 shadow-lg">
<CardHeader>
<CardTitle className="flex items-center gap-2">
@@ -729,8 +735,9 @@ export default function NewReel() {
</Button>
</CardContent>
</Card>
)}
{generatedVariants.length > 0 && (
{!srtFile && generatedVariants.length > 0 && (
<Card className="rounded-2xl border-border/50 shadow-lg">
<CardHeader>
<CardTitle>Variations générées</CardTitle>
@@ -770,16 +777,26 @@ export default function NewReel() {
Texte Overlay
</CardTitle>
<CardDescription>
Ce texte s'affichera au centre de votre Reel (style TikTok)
Ce texte s'affichera au centre de votre Reel (style TikTok).
Vous pouvez aussi importer un fichier SRT : il remplace le texte.
</CardDescription>
</CardHeader>
<CardContent>
<Textarea
value={overlayText}
onChange={(e) => setOverlayText(e.target.value)}
placeholder="Écrivez le texte qui apparaîtra sur votre Reel..."
rows={4}
/>
<div className="mb-4">
<SrtUpload value={srtFile} onChange={setSrtFile} />
</div>
{srtFile ? (
<p className="text-xs text-muted-foreground">
Texte libre désactivé : le fichier SRT est utilisé. Retirez-le pour écrire un texte.
</p>
) : (
<Textarea
value={overlayText}
onChange={(e) => setOverlayText(e.target.value)}
placeholder="Écrivez le texte qui apparaîtra sur votre Reel..."
rows={4}
/>
)}
<div className="flex items-center space-x-2 mt-4">
<Switch
@@ -834,12 +851,12 @@ export default function NewReel() {
<VoicePicker
value={voiceSettings}
onChange={setVoiceSettings}
sampleText={overlayText}
sampleText={srtFile ? srtFile.cues[0]?.text : overlayText}
onPreview={setVoicePreview}
/>
</div>
{syncInfo && (
{syncInfo && !srtFile && (
<div className={`p-3 rounded-lg border mt-3 ${syncInfo.isHealthy ? 'bg-green-500/10 border-green-500/30' : 'bg-yellow-500/10 border-yellow-500/30'}`}>
<div className="flex items-center justify-between text-sm">
<span className="font-medium">Sync texte/voix</span>
@@ -857,8 +874,10 @@ export default function NewReel() {
)}
<p className="text-xs text-muted-foreground mt-2">
Le texte sera automatiquement synchronisé avec la voix.
Les #hashtags et émojis ne seront pas lus.
{srtFile
? "Chaque sous-titre sera lu à son instant et accéléré si besoin pour respecter la durée du fichier SRT."
: "Le texte sera automatiquement synchronisé avec la voix."}
{' '}Les #hashtags et émojis ne seront pas lus.
</p>
</div>
)}
@@ -984,7 +1003,8 @@ export default function NewReel() {
<ReelPreview
kind="video"
videoUrl={selectedVideo.originalUrl}
text={overlayText}
text={srtFile ? srtText(srtFile.cues) : overlayText}
srtCues={srtFile?.cues}
showCaptions={drawText}
captionStyle={captionStyle}
ttsEnabled={ttsEnabled}
@@ -1022,7 +1042,7 @@ export default function NewReel() {
</div>
<div className="flex justify-between">
<span className="text-muted-foreground">Texte</span>
<span>{overlayText ? '✓' : '—'}</span>
<span>{srtFile ? `SRT (${srtFile.cues.length})` : overlayText ? '✓' : '—'}</span>
</div>
<div className="flex justify-between">
<span className="text-muted-foreground">Pages</span>
+6
View File
@@ -90,6 +90,12 @@ def align_words(display_text: str, spoken: list[Word], total_duration: float | N
return words
def spread_words(text: str, start: float, end: float) -> list[Word]:
"""Mots d'un texte répartis sur un intervalle (sous-titre SRT sans voix)."""
tokens = display_tokens(text)
return _spread(tokens, start, end) if tokens and end > start else []
def _spread(tokens: list[str], start: float, end: float) -> list[Word]:
"""Répartit des mots sur un intervalle au prorata de leur longueur."""
weights = [max(1, len(normalize(t))) for t in tokens]
+54 -3
View File
@@ -4,6 +4,7 @@ import asyncio
import base64
import contextlib
import logging
import shutil
import time
from pathlib import Path
@@ -12,7 +13,7 @@ from fastapi import Depends, FastAPI, Header, HTTPException
from fastapi.responses import FileResponse
from pydantic import BaseModel
from . import align, config, jobs, proc, render, subtitles, tts, voices
from . import align, config, jobs, proc, render, srt_voice, subtitles, tts, voices
from .audio import encode_preview
from .text import clean_text
@@ -57,10 +58,18 @@ class TtsRequest(BaseModel):
gemini_api_key: str | None = None
class SrtCue(BaseModel):
start: float
end: float
text: str
class ReelRequest(BaseModel):
video_base64: str | None = None
video_url: str | None = None
text: str | None = None
# Sous-titres SRT : remplacent `text`, la voix lit chacun à son instant
srt_cues: list[SrtCue] | None = None
music_url: str | None = None
watermark_url: str | None = None
store_name: str | None = None
@@ -182,6 +191,31 @@ async def _synthesize(text: str, display_source: str | None, request, workdir: P
raise HTTPException(status_code=502, detail=f"La voix n'a pas pu être générée : {error}") from error
def _srt_cues(request: ReelRequest) -> list[srt_voice.Cue]:
"""Sous-titres lisibles du SRT, dans l'ordre (vide sans SRT)."""
cues = [
srt_voice.Cue(c.start, c.end, c.text)
for c in request.srt_cues or []
if c.end > c.start and clean_text(c.text)
]
return sorted(cues, key=lambda c: c.start)
async def _synthesize_srt(cues: list[srt_voice.Cue], request, workdir: Path) -> tts.VoiceTrack:
try:
return await srt_voice.synthesize_cues(
cues=cues,
engine=request.tts_engine,
voice=request.tts_voice,
style=request.tts_style,
gemini_api_key=request.gemini_api_key,
workdir=workdir,
)
except Exception as error:
log.exception("Voix SRT impossible à générer")
raise HTTPException(status_code=502, detail=f"La voix n'a pas pu être générée : {error}") from error
async def _download(url: str, target: Path, what: str, required: bool) -> bool:
try:
async with httpx.AsyncClient(timeout=httpx.Timeout(120, connect=15), follow_redirects=True) as client:
@@ -240,8 +274,11 @@ async def _gather(request: ReelRequest, workdir: Path, clock: Stopwatch, *, fetc
clock.lap("download")
track = None
cues = _srt_cues(request)
spoken_text = clean_text(request.text)
if request.tts_enabled and spoken_text:
if request.tts_enabled and cues:
track = await _synthesize_srt(cues, request, workdir)
elif request.tts_enabled and spoken_text:
track = await _synthesize(spoken_text, request.text, request, workdir)
clock.lap("tts")
@@ -254,6 +291,8 @@ async def _gather(request: ReelRequest, workdir: Path, clock: Stopwatch, *, fetc
music_volume=request.music_volume,
voice=track.path if track else None,
voice_duration=track.duration if track else 0.0,
# Avec un SRT, la voix est déjà posée aux instants du fichier
voice_delay=0.0 if cues else config.VOICE_DELAY,
watermark=watermark if has_watermark else None,
outro_expected=not fetch_logo and request.has_logo,
ending_effect=request.enable_ending_effect,
@@ -273,6 +312,14 @@ async def _gather(request: ReelRequest, workdir: Path, clock: Stopwatch, *, fetc
async def _caption_words(request: ReelRequest, plan: render.RenderPlan, track, info) -> list[align.Word]:
"""Mots à afficher, en secondes depuis le début de la vidéo."""
cues = _srt_cues(request)
if cues:
if not request.draw_text:
return []
if track:
return track.words
# Sans voix : chaque sous-titre s'affiche sur son propre intervalle
return [w for c in cues for w in align.spread_words(clean_text(c.text), c.start, c.end)]
display = clean_text(request.text)
if not request.draw_text or not display:
return []
@@ -284,7 +331,11 @@ async def _caption_words(request: ReelRequest, plan: render.RenderPlan, track, i
def _keep_only(workdir: Path, keep: set[Path]) -> None:
"""Seuls les fichiers à télécharger restent jusqu'à la récupération."""
for entry in workdir.iterdir():
if entry not in keep:
if entry in keep:
continue
if entry.is_dir():
shutil.rmtree(entry, ignore_errors=True) # voix SRT : un dossier par sous-titre
else:
entry.unlink(missing_ok=True)
+183
View File
@@ -0,0 +1,183 @@
"""Voix calée sur un fichier SRT.
Chaque sous-titre est lu séparément puis posé à son instant de début. Si la
lecture est plus longue que le sous-titre, elle est accélérée (sans changer
la hauteur de la voix) pour finir à temps : la voix respecte le minutage du
fichier, au lieu d'un texte lu d'une traite.
"""
import logging
from dataclasses import dataclass
from pathlib import Path
from . import proc, tts
from .align import Word
from .text import clean_text
log = logging.getLogger(__name__)
# Débordement toléré sur le silence qui suit un sous-titre avant d'accélérer
OVERFLOW_TOLERANCE = 0.25
# Marges gardées autour des mots lors du retrait des silences de la voix
TRIM_LEAD = 0.05
TRIM_TAIL = 0.12
# Au-delà, l'accélération s'entend nettement : l'utilisateur est prévenu
AUDIBLE_SPEEDUP = 1.35
@dataclass
class Cue:
start: float
end: float
text: str
def cue_windows(cues: list[Cue]) -> list[float]:
"""Durée disponible pour lire chaque sous-titre sans empiéter sur le suivant."""
windows = []
for i, cue in enumerate(cues):
limit = cue.end + OVERFLOW_TOLERANCE
if i + 1 < len(cues):
limit = min(limit, max(cue.end, cues[i + 1].start))
windows.append(max(0.1, limit - cue.start))
return windows
def speed_factor(duration: float, window: float) -> float:
"""Accélération nécessaire pour tenir dans la fenêtre (1 = vitesse normale)."""
return max(1.0, duration / window) if window > 0 else 1.0
def atempo_chain(factor: float) -> str:
"""Filtre atempo ; découpé en étapes ≤ 2 pour les anciennes versions de FFmpeg."""
steps = []
remaining = factor
while remaining > 2.0:
steps.append(2.0)
remaining /= 2.0
steps.append(remaining)
return ",".join(f"atempo={s:.4f}" for s in steps)
def spoken_span(words: list[Word], duration: float) -> tuple[float, float]:
"""Partie parlée de la voix : les silences de début et de fin sont retirés."""
if not words:
return 0.0, duration
start = max(0.0, words[0].start - TRIM_LEAD)
end = min(duration, words[-1].end + TRIM_TAIL)
return (start, end) if end > start else (0.0, duration)
async def synthesize_cues(
*,
cues: list[Cue],
engine: str | None,
voice: str | None,
style: str | None,
gemini_api_key: str | None,
workdir: Path,
) -> tts.VoiceTrack:
"""Voix complète : mots minutés en secondes depuis le début de la vidéo."""
cues = sorted((c for c in cues if clean_text(c.text)), key=lambda c: c.start)
if not cues:
raise ValueError("Aucun sous-titre lisible dans le fichier SRT")
warnings: list[str] = []
windows = cue_windows(cues)
segments: list[tuple[Path, float]] = []
words: list[Word] = []
used_engine, used_voice = engine or "gemini", voice or ""
for index, (cue, window) in enumerate(zip(cues, windows)):
cue_dir = workdir / f"cue_{index:03d}"
cue_dir.mkdir(exist_ok=True)
text = clean_text(cue.text)
track = await tts.synthesize(
text=text,
display_text=text,
engine=used_engine,
voice=voice,
style=style,
gemini_api_key=gemini_api_key,
workdir=cue_dir,
)
for warning in track.warnings:
if warning not in warnings:
warnings.append(warning)
# Après un repli sur Edge, on reste sur Edge : une seule voix du début à la fin
used_engine, used_voice = track.engine, track.voice
span_start, span_end = spoken_span(track.words, track.duration)
factor = speed_factor(span_end - span_start, window)
if factor > AUDIBLE_SPEEDUP:
warnings.append(
f"Sous-titre {index + 1} (« {cue.text[:40]} ») lu {factor:.1f}× plus vite "
"pour tenir dans son minutage : raccourcissez-le ou allongez sa durée."
)
fitted = cue_dir / "fitted.wav"
trim = f"atrim=start={span_start:.3f}:end={span_end:.3f}"
await proc.run(
[
"ffmpeg",
"-y",
"-hide_banner",
"-loglevel",
"error",
"-i",
str(track.path),
"-af",
f"{trim},asetpts=PTS-STARTPTS,{atempo_chain(factor)}",
"-ac",
"1",
"-ar",
"48000",
"-c:a",
"pcm_s16le",
str(fitted),
],
timeout=120,
)
segments.append((fitted, cue.start))
for w in track.words:
start = cue.start + max(0.0, w.start - span_start) / factor
end = cue.start + max(0.0, w.end - span_start) / factor
words.append(Word(w.text, start, min(max(end, start + 0.05), cue.start + window)))
output = workdir / "voice.wav"
await proc.run(mix_command(segments, output), timeout=300)
duration = (await proc.probe(output)).duration
log.info("Voix SRT prête : %s/%s, %d sous-titres, %.1f s", used_engine, used_voice, len(cues), duration)
return tts.VoiceTrack(output, duration, _monotonic(words), used_engine, used_voice, warnings)
def mix_command(segments: list[tuple[Path, float]], output: Path) -> list[str]:
"""Pose chaque segment de voix à son instant, sur une seule piste."""
command = ["ffmpeg", "-y", "-hide_banner", "-loglevel", "error"]
for path, _ in segments:
command += ["-i", str(path)]
graph = [f"[{i}:a]adelay={int(round(start * 1000))}:all=1[s{i}]" for i, (_, start) in enumerate(segments)]
labels = "".join(f"[s{i}]" for i in range(len(segments)))
graph.append(
f"{labels}amix=inputs={len(segments)}:duration=longest:dropout_transition=0:normalize=0[voice]"
)
return command + [
"-filter_complex",
";".join(graph),
"-map",
"[voice]",
"-ac",
"1",
"-ar",
"48000",
"-c:a",
"pcm_s16le",
str(output),
]
def _monotonic(words: list[Word]) -> list[Word]:
for prev, cur in zip(words, words[1:]):
cur.start = max(cur.start, prev.start + 0.01)
cur.end = max(cur.end, cur.start + 0.05)
return words
+51
View File
@@ -0,0 +1,51 @@
from pathlib import Path
from app.align import Word, spread_words
from app.render import RenderPlan
from app.srt_voice import Cue, atempo_chain, cue_windows, mix_command, speed_factor, spoken_span
def test_window_stops_at_next_cue_and_tolerates_a_short_overflow():
cues = [Cue(0.0, 2.0, "a"), Cue(2.1, 4.0, "b"), Cue(6.0, 7.0, "c")]
assert cue_windows(cues) == [2.1, 2.15, 1.25]
def test_voice_is_sped_up_only_when_too_long():
assert speed_factor(1.5, 2.0) == 1.0
assert speed_factor(3.0, 2.0) == 1.5
def test_atempo_is_split_into_supported_steps():
assert atempo_chain(1.25) == "atempo=1.2500"
assert atempo_chain(3.0) == "atempo=2.0000,atempo=1.5000"
def test_spoken_span_trims_silences_around_words():
words = [Word("Bonjour", 0.4, 0.9), Word("!", 0.9, 1.2)]
assert [round(t, 3) for t in spoken_span(words, 2.0)] == [0.35, 1.32]
assert spoken_span([], 2.0) == (0.0, 2.0)
def test_each_segment_is_placed_at_its_cue_start():
cmd = mix_command([(Path("a.wav"), 0.0), (Path("b.wav"), 3.25)], Path("voice.wav"))
graph = cmd[cmd.index("-filter_complex") + 1]
assert "[1:a]adelay=3250:all=1[s1]" in graph
assert "amix=inputs=2" in graph and "normalize=0" in graph
def test_srt_voice_starts_without_delay():
plan = RenderPlan(
video=Path("in.mp4"),
video_duration=5.0,
output=Path("out.mp4"),
voice=Path("v.wav"),
voice_duration=8.0,
voice_delay=0.0,
)
assert plan.speech_end == 8.0
assert plan.total_duration == 8.8
def test_words_spread_over_their_cue():
words = spread_words("Bonjour à tous", 4.0, 5.0)
assert words[0].start == 4.0 and round(words[-1].end, 3) == 5.0
+5
View File
@@ -7,6 +7,7 @@ import type { User } from '@shared/schema';
import { storage } from '../storage';
import { ffmpegService, FFmpegServiceError } from '../services/ffmpeg';
import { ttsPreviewSchema, videoReelParamsSchema, type VideoReelParams } from '@shared/reel';
import { srtText } from '@shared/srt';
import { enqueueReelJob, countActiveReelJobs } from '../services/reels/queue';
import { resolveGeminiApiKey, resolveLogoPath, resolveStoreName } from '../services/reels/assets';
import { configuredRenderer } from '../services/reels/videoPipeline';
@@ -334,8 +335,12 @@ reelsRouter.post('/reels', async (req: Request, res: Response) => {
return res.status(400).json({ error: parsed.error.issues[0]?.message ?? 'Paramètres invalides' });
}
const { srtCues } = parsed.data;
const params: VideoReelParams = {
...parsed.data,
// Avec un SRT, le texte libre n'est plus utilisé : la voix lit les sous-titres
overlayText: srtCues ? undefined : parsed.data.overlayText,
description: parsed.data.description || (srtCues ? srtText(srtCues) : undefined),
storeName: await resolveStoreName(user.id, parsed.data.pageIds[0]),
};
+5
View File
@@ -10,6 +10,7 @@ import fs from 'fs';
import { Readable } from 'stream';
import { pipeline } from 'stream/promises';
import type { ReadableStream as WebReadableStream } from 'stream/web';
import type { SrtCue } from '@shared/srt';
import type { TtsEngine, TtsStyle } from '@shared/voices';
/** Un rendu long (stabilisation + encodage) peut dépasser plusieurs minutes. */
@@ -20,6 +21,8 @@ const HEALTH_TIMEOUT_MS = 5_000;
export interface ReelRenderOptions {
text?: string;
/** Sous-titres SRT : remplacent `text`, la voix lit chacun à son instant. */
srtCues?: SrtCue[];
musicUrl?: string;
ttsEnabled?: boolean;
ttsVoice?: string;
@@ -126,6 +129,7 @@ export class FFmpegService {
const body = {
video_url: videoUrl,
text: options.text,
srt_cues: options.srtCues,
music_url: options.musicUrl,
tts_enabled: options.ttsEnabled ?? false,
tts_voice: options.ttsVoice,
@@ -196,6 +200,7 @@ export class FFmpegService {
body: JSON.stringify({
video_url: videoUrl,
text: options.text,
srt_cues: options.srtCues,
music_url: options.musicUrl,
tts_enabled: options.ttsEnabled ?? false,
tts_voice: options.ttsVoice,
+2 -1
View File
@@ -47,7 +47,8 @@ export async function runVideoReelJob({ job, progress }: JobContext) {
]);
const options: ReelRenderOptions = {
text: params.overlayText,
text: params.srtCues ? undefined : params.overlayText,
srtCues: params.srtCues,
musicUrl,
ttsEnabled: params.ttsEnabled,
ttsVoice: params.ttsVoice,
+3 -1
View File
@@ -46,9 +46,11 @@ export function computeReelTiming(input: {
voiceDuration?: number;
hasOutro: boolean;
endingEffect: boolean;
/** 0 pour une voix déjà posée aux instants d'un fichier SRT. */
voiceDelay?: number;
}): ReelTiming {
const hasOutro = input.hasOutro && input.endingEffect;
const speechEnd = input.voiceDuration ? VOICE_DELAY + input.voiceDuration : 0;
const speechEnd = input.voiceDuration ? (input.voiceDelay ?? VOICE_DELAY) + input.voiceDuration : 0;
let total = input.videoDuration;
if (input.voiceDuration) {
total = Math.max(total, speechEnd + VOICE_TAIL);
+3
View File
@@ -6,6 +6,7 @@
import { z } from "zod";
import { CAPTION_STYLES, DEFAULT_CAPTION_STYLE } from "./captions";
import { srtCuesSchema } from "./srt";
import { TTS_ENGINES, TTS_STYLES } from "./voices";
const optionalText = z
@@ -23,6 +24,8 @@ export const videoReelParamsSchema = z.object({
musicTrackId: optionalText,
musicUrl: optionalText,
overlayText: optionalText,
// Sous-titres SRT : remplacent le texte libre, la voix lit chacun à son instant
srtCues: srtCuesSchema.optional(),
description: optionalText,
ttsEnabled: z.boolean().default(false),
ttsVoice: optionalText,
+54
View File
@@ -0,0 +1,54 @@
import { describe, expect, it } from "vitest";
import { parseSrt, srtEnd, srtText, srtWords } from "./srt";
import { videoReelParamsSchema } from "./reel";
const SAMPLE = `1
00:00:01,000 --> 00:00:03,500
Bonjour à tous !
2
00:00:04,000 --> 00:00:06,250
<i>Découvrez</i> nos
nouveautés en magasin
3
00:00:07,000 --> 00:00:07,000
Durée nulle, ignoré
`;
describe("parseSrt", () => {
it("lit les sous-titres, retire les balises et réunit les lignes", () => {
const cues = parseSrt(SAMPLE.replace(/\n/g, "\r\n"));
expect(cues).toEqual([
{ start: 1, end: 3.5, text: "Bonjour à tous !" },
{ start: 4, end: 6.25, text: "Découvrez nos nouveautés en magasin" },
]);
expect(srtText(cues)).toBe("Bonjour à tous ! Découvrez nos nouveautés en magasin");
expect(srtEnd(cues)).toBe(6.25);
});
it("accepte les millisecondes séparées par un point et sans numéro", () => {
expect(parseSrt("00:00:02.5 --> 00:00:04.0\nSalut")).toEqual([{ start: 2.5, end: 4, text: "Salut" }]);
});
it("refuse un fichier sans sous-titre", () => {
expect(() => parseSrt("pas un srt")).toThrow(/Aucun sous-titre/);
});
it("répartit les mots dans l'intervalle de leur sous-titre", () => {
const words = srtWords(parseSrt(SAMPLE));
const second = words.findIndex((w) => w.text === "Découvrez");
expect(words[0].start).toBe(1);
expect(words[second - 1].end).toBeCloseTo(3.5);
expect(words[second].start).toBe(4);
expect(words[words.length - 1].end).toBeCloseTo(6.25);
});
});
describe("videoReelParamsSchema avec SRT", () => {
it("accepte des sous-titres valides et refuse une durée nulle", () => {
const base = { videoMediaId: "v", pageIds: ["p"] };
expect(videoReelParamsSchema.safeParse({ ...base, srtCues: [{ start: 0, end: 1, text: "a" }] }).success).toBe(true);
expect(videoReelParamsSchema.safeParse({ ...base, srtCues: [{ start: 1, end: 1, text: "a" }] }).success).toBe(false);
});
});
+94
View File
@@ -0,0 +1,94 @@
/**
* Fichiers de sous-titres SRT : lecture côté navigateur, validation côté
* serveur et minutage de l'aperçu. Quand un SRT est fourni, il remplace le
* texte libre : la voix lit chaque sous-titre à son instant.
*/
import { z } from "zod";
import { cleanCaptionText, spreadWords, type TimedWord } from "./captions";
export const MAX_SRT_CUES = 300;
export const srtCueSchema = z
.object({
start: z.number().min(0),
end: z.number().min(0),
text: z.string().trim().min(1).max(500),
})
.refine((cue) => cue.end > cue.start, { message: "Sous-titre de durée nulle" });
export const srtCuesSchema = z
.array(srtCueSchema)
.min(1, "Fichier SRT vide")
.max(MAX_SRT_CUES, `Fichier SRT trop long (${MAX_SRT_CUES} sous-titres au plus)`);
export type SrtCue = z.infer<typeof srtCueSchema>;
const TIMING = /(\d{1,2}):(\d{2}):(\d{2})[,.](\d{1,3})\s*-->\s*(\d{1,2}):(\d{2}):(\d{2})[,.](\d{1,3})/;
function toSeconds(h: string, m: string, s: string, ms: string): number {
return Number(h) * 3600 + Number(m) * 60 + Number(s) + Number(ms.padEnd(3, "0")) / 1000;
}
/**
* Lit un fichier SRT. Les balises (<i>, {\an8}…) sont retirées, les lignes
* d'un même sous-titre réunies ; les sous-titres sans texte lisible sont
* ignorés. Lève une erreur lisible si aucun sous-titre n'est trouvé.
*/
export function parseSrt(content: string): SrtCue[] {
const blocks = content
.replace(/^/, "")
.replace(/\r\n?/g, "\n")
.split(/\n\s*\n/);
const cues: SrtCue[] = [];
for (const block of blocks) {
const lines = block.split("\n").map((l) => l.trim()).filter(Boolean);
const timingIndex = lines.findIndex((l) => TIMING.test(l));
if (timingIndex < 0) continue;
const match = lines[timingIndex].match(TIMING)!;
const start = toSeconds(match[1], match[2], match[3], match[4]);
const end = toSeconds(match[5], match[6], match[7], match[8]);
const text = lines
.slice(timingIndex + 1)
.join(" ")
.replace(/<[^>]+>/g, "")
.replace(/\{[^}]*\}/g, "")
.replace(/\s+/g, " ")
.trim();
if (!text || end <= start || !cleanCaptionText(text)) continue;
cues.push({ start: round(start), end: round(end), text });
}
if (!cues.length) throw new Error("Aucun sous-titre valide trouvé dans ce fichier SRT");
if (cues.length > MAX_SRT_CUES) {
throw new Error(`Fichier SRT trop long (${MAX_SRT_CUES} sous-titres au plus)`);
}
return cues.sort((a, b) => a.start - b.start);
}
function round(seconds: number): number {
return Math.round(seconds * 1000) / 1000;
}
/** Texte complet du SRT (description de la publication, aperçu de la voix). */
export function srtText(cues: SrtCue[]): string {
return cues.map((c) => c.text).join(" ");
}
/** Fin du dernier sous-titre : la vidéo dure au moins jusque-là. */
export function srtEnd(cues: SrtCue[]): number {
return cues.reduce((max, c) => Math.max(max, c.end), 0);
}
/** Mots de chaque sous-titre répartis sur son intervalle (aperçu, rendu sans voix). */
export function srtWords(cues: SrtCue[]): TimedWord[] {
return cues.flatMap((c) => spreadWords(c.text, c.start, c.end));
}
/** « 1:05 » pour l'affichage. */
export function formatSrtTime(seconds: number): string {
const m = Math.floor(seconds / 60);
const s = Math.floor(seconds % 60);
return `${m}:${String(s).padStart(2, "0")}`;
}