feat(reels): voix Gemini complète, sous-titres mot à mot et rendu de qualité

Service ffmpeg-api réécrit (ffmpeg-service/app/) :
- Appels FFmpeg asynchrones : le service ne se fige plus pendant un rendu
- Vidéo récupérée par téléchargement (GET /files/…) au lieu de base64 en JSON
- Vraies erreurs HTTP ; échec explicite si la voix demandée est impossible
- 30 voix Gemini + ton de lecture (dynamique, chaleureux, promo, calme),
  clé en en-tête, modèle configurable avec repli
- Edge TTS 7 (minutage des mots restauré), secours en voix françaises
- Voix traitée : filtre, compression, niveau constant ; musique bouclée et
  baissée automatiquement sous la voix ; mix final à -14 LUFS
- Sous-titres calés mot à mot (Whisper pour Gemini et la voix d'origine,
  à la place de ffsubsync), style Montserrat, placés hors des boutons Reels
- Vidéo : plus de retouche luminosité forcée, scaling lanczos, HDR iPhone
  converti, BT.709, AAC 48 kHz 192k ; la vidéo s'allonge si la voix dépasse
- Grand logo de fin affiché après la voix ; FFmpeg 7.0.2 épinglé, polices et
  modèle Whisper intégrés à l'image ; tests pytest et ruff

Application :
- Sélecteur de voix partagé (4 pages) : moteur, 30 voix, ton, écoute
- Reel images : minutage réel des mots, interrupteur voix respecté
- sync-info ne génère plus de voix à chaque frappe (estimation locale)
- Stabilisation désactivée par défaut, route /reels/preview inutilisée retirée
- Log « [ReelQueue] Worker démarré » pour vérifier la version déployée

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_018Ze4bs7tpF1KGWUk6ZZSZ4
This commit is contained in:
Claude committed 2026-09-24 12:31:27 +00:00
1 parent d24ab0d4b7
commit 3c6a66bf64
47 files changed
+2206 -2285

No files matched your search

+19
View File
@@ -105,3 +105,22 @@ LEGAL_CONTACT_EMAIL=
# Clé d'accès pour l'API externe /api/v1/*
# Générer avec: openssl rand -hex 32
EXTERNAL_API_KEY=your-secure-random-key-here
# ==========================================
# REELS : VOIX ET RENDU VIDÉO (service ffmpeg-api)
# ==========================================
# Clé partagée entre l'application et le service FFmpeg (CHANGEZ CETTE VALEUR).
FFMPEG_API_KEY=change-me-ffmpeg-key
#
# Clé Google AI Studio pour les voix Gemini (sinon : voix Edge, gratuite).
# Peut aussi être saisie dans l'application, ce qui est prioritaire.
GEMINI_API_KEY=
#
# Modèle Gemini TTS (défaut : gemini-2.5-flash-preview-tts ; repli automatique
# sur celui-ci si le modèle choisi n'existe pas).
# GEMINI_TTS_MODEL=
#
# Modèle Whisper utilisé pour caler les sous-titres mot à mot sur la voix
# (tiny, base, small). Choisi à la construction de l'image : changer la valeur
# demande de reconstruire ffmpeg-api.
# WHISPER_MODEL=base
+14
View File
@@ -174,6 +174,20 @@ La stack se construit depuis les sources : `app` et `ffmpeg-api` ont une section
`socialflow-ffmpeg-api:latest`) : il ne dépend donc plus du nom donné à la
stack dans Portainer.
## 🎙️ Service ffmpeg-api (voix et rendu des Reels)
- **Premier build plus long** : l'image embarque FFmpeg 7.0.2 (version
épinglée), les polices des sous-titres et le modèle Whisper qui cale les
sous-titres mot à mot sur la voix (~150 Mo avec `WHISPER_MODEL=base`).
- **Voix Gemini** : renseignez `GEMINI_API_KEY` (ou la clé dans l'application).
Sans clé, la voix Edge gratuite est utilisée et un avertissement apparaît
dans les logs.
- **Vérifier la version déployée** : au démarrage, les logs de `socialflow-app`
affichent `[ReelQueue] Worker démarré`, et `socialflow-ffmpeg` répond
`{"status":"ok","version":2}` sur `/health`.
- **Tests du service** : `pip install -r requirements-dev.txt`, puis `pytest`
et `ruff check .` dans `ffmpeg-service/`.
## 🔒 Sécurité en production
1. **Variables d'environnement** : Ne commitez JAMAIS le fichier `.env`
@@ -0,0 +1,185 @@
import { useEffect, useRef, useState } from "react";
import { Loader2, Play, Square } from "lucide-react";
import { Button } from "@/components/ui/button";
import { Label } from "@/components/ui/label";
import {
Select,
SelectContent,
SelectGroup,
SelectItem,
SelectLabel,
SelectTrigger,
SelectValue,
} from "@/components/ui/select";
import { useToast } from "@/hooks/use-toast";
import { apiRequest, getErrorMessage } from "@/lib/queryClient";
import {
DEFAULT_VOICE,
TTS_STYLE_OPTIONS,
voicesFor,
type TtsEngine,
type TtsStyle,
} from "@shared/voices";
export interface VoiceSettings {
engine: TtsEngine;
voice: string;
style: TtsStyle;
}
interface VoicePickerProps {
value: VoiceSettings;
onChange: (value: VoiceSettings) => void;
/** Texte lu par le bouton « Tester la voix ». */
sampleText?: string;
compact?: boolean;
}
const FALLBACK_SAMPLE = "Découvrez nos nouveautés en magasin, on vous attend !";
/**
* Choix du moteur, de la voix et du ton, avec écoute de l'aperçu.
* Partagé par les pages Reel (vidéo et images, bureau et mobile).
*/
export function VoicePicker({ value, onChange, sampleText, compact = false }: VoicePickerProps) {
const { toast } = useToast();
const [loading, setLoading] = useState(false);
const [playing, setPlaying] = useState(false);
const audioRef = useRef<HTMLAudioElement | null>(null);
useEffect(() => () => audioRef.current?.pause(), []);
const voices = voicesFor(value.engine);
const groups = [
{ label: "Voix féminines", items: voices.filter((v) => v.gender === "female") },
{ label: "Voix masculines", items: voices.filter((v) => v.gender === "male") },
];
const setEngine = (engine: TtsEngine) => {
if (engine !== value.engine) onChange({ ...value, engine, voice: DEFAULT_VOICE[engine] });
};
const stop = () => {
audioRef.current?.pause();
setPlaying(false);
};
const preview = async () => {
if (playing) return stop();
setLoading(true);
try {
const response = await apiRequest("POST", "/api/reels/tts-preview", {
text: sampleText?.trim() || FALLBACK_SAMPLE,
ttsEngine: value.engine,
ttsVoice: value.voice,
ttsStyle: value.style,
});
const data = await response.json();
for (const warning of data.warnings ?? []) {
toast({ title: "Voix de secours utilisée", description: warning });
}
const audio = new Audio(`data:audio/mpeg;base64,${data.audioBase64}`);
audioRef.current?.pause();
audioRef.current = audio;
audio.onended = () => setPlaying(false);
await audio.play();
setPlaying(true);
} catch (error) {
toast({
title: "Impossible de tester la voix",
description: getErrorMessage(error, "La voix n'a pas pu être générée."),
variant: "destructive",
});
} finally {
setLoading(false);
}
};
const labelClass = compact ? "text-xs font-medium" : "text-sm font-medium";
return (
<div className="space-y-3">
<div className="flex items-center gap-3">
<Label className={`${labelClass} w-14 shrink-0`}>Moteur</Label>
<div className="flex gap-2">
<Button
type="button"
size="sm"
className={compact ? "h-7 text-xs" : undefined}
variant={value.engine === "gemini" ? "default" : "outline"}
onClick={() => setEngine("gemini")}
>
Gemini (naturelle)
</Button>
<Button
type="button"
size="sm"
className={compact ? "h-7 text-xs" : undefined}
variant={value.engine === "edge" ? "default" : "outline"}
onClick={() => setEngine("edge")}
>
Edge (gratuite)
</Button>
</div>
</div>
<div className="flex items-center gap-3">
<Label className={`${labelClass} w-14 shrink-0`}>Voix</Label>
<Select value={value.voice} onValueChange={(voice) => onChange({ ...value, voice })}>
<SelectTrigger className={compact ? "h-8 text-xs flex-1" : "flex-1"}>
<SelectValue placeholder="Choisir une voix" />
</SelectTrigger>
<SelectContent className="max-h-72">
{groups.map((group) => (
<SelectGroup key={group.label}>
<SelectLabel>{group.label}</SelectLabel>
{group.items.map((voice) => (
<SelectItem key={voice.id} value={voice.id}>
{voice.label}
</SelectItem>
))}
</SelectGroup>
))}
</SelectContent>
</Select>
</div>
<div className="flex items-center gap-3">
<Label className={`${labelClass} w-14 shrink-0`}>Ton</Label>
<Select value={value.style} onValueChange={(style) => onChange({ ...value, style: style as TtsStyle })}>
<SelectTrigger className={compact ? "h-8 text-xs flex-1" : "flex-1"}>
<SelectValue />
</SelectTrigger>
<SelectContent>
{TTS_STYLE_OPTIONS.map((style) => (
<SelectItem key={style.id} value={style.id}>
{style.label} <span className="text-muted-foreground">— {style.description}</span>
</SelectItem>
))}
</SelectContent>
</Select>
</div>
<Button
type="button"
size="sm"
variant="secondary"
className={compact ? "w-full h-8 text-xs" : "w-full"}
disabled={loading}
onClick={(e) => {
e.stopPropagation();
preview();
}}
>
{loading ? (
<Loader2 className="w-3 h-3 mr-2 animate-spin" />
) : playing ? (
<Square className="w-3 h-3 mr-2" />
) : (
<Play className="w-3 h-3 mr-2" />
)}
{loading ? "Génération de la voix…" : playing ? "Arrêter" : "Tester la voix"}
</Button>
</div>
);
}
+17 -93
View File
@@ -16,6 +16,8 @@ import { Switch } from "@/components/ui/switch";
import { Label } from "@/components/ui/label";
import { Slider } from "@/components/ui/slider";
import { useToast } from "@/hooks/use-toast";
import { VoicePicker, type VoiceSettings } from "@/components/reels/voice-picker";
import { DEFAULT_TTS_STYLE, DEFAULT_VOICE } from "@shared/voices";
import { apiRequest, queryClient, handleUnauthorized } from "@/lib/queryClient";
import type { SocialPage, Media } from "@shared/schema";
import { SiFacebook, SiTiktok } from "react-icons/si";
@@ -59,27 +61,17 @@ export default function MobileNewReel() {
// État TTS
const [ttsEnabled, setTtsEnabled] = useState(false);
const [ttsEngine, setTtsEngine] = useState<'edge' | 'gemini'>('edge');
const [ttsVoice, setTtsVoice] = useState('fr-FR-VivienneMultilingualNeural');
// Gemini native TTS voices (Charon = homme, Kore = femme)
const geminiVoices = [
{ label: 'Charon - Voix Homme', value: 'fr-FR-Standard-B' },
{ label: 'Kore - Voix Femme', value: 'fr-FR-Standard-A' },
];
// French Edge TTS voices
const edgeVoices = [
{ label: 'Vivienne (Femme)', value: 'fr-FR-VivienneMultilingualNeural' },
{ label: 'Henri (Homme)', value: 'fr-FR-HenriNeural' },
{ label: 'Denise (Femme)', value: 'fr-FR-DeniseNeural' },
{ label: 'Rémy (Homme)', value: 'fr-FR-RemyMultilingualNeural' },
{ label: 'Jenny (Anglaise, Femme)', value: 'en-US-JennyNeural' },
];
const [voiceSettings, setVoiceSettings] = useState<VoiceSettings>({
engine: 'gemini',
voice: DEFAULT_VOICE.gemini,
style: DEFAULT_TTS_STYLE,
});
const { engine: ttsEngine, voice: ttsVoice, style: ttsStyle } = voiceSettings;
// État audio preview
const [isPlaying, setIsPlaying] = useState<string | null>(null);
const [stabilize, setStabilize] = useState(true); // Activé par défaut pour les Reels
// Désactivée par défaut : double le temps de rendu, utile seulement pour une vidéo tremblée
const [stabilize, setStabilize] = useState(false);
const [enableEndingEffect, setEnableEndingEffect] = useState(true);
// TTS Sync state
@@ -271,6 +263,7 @@ export default function MobileNewReel() {
ttsEnabled,
ttsEngine,
ttsVoice,
ttsStyle,
enableEndingEffect,
});
};
@@ -492,82 +485,13 @@ export default function MobileNewReel() {
<span>TTS — voix activée</span>
</div>
{/* TTS Engine Selector */}
<div className="flex items-center gap-2 mt-2">
<Label className="text-xs font-medium">Moteur:</Label>
<div className="flex gap-1">
<Button
size="sm"
variant={ttsEngine === 'edge' ? 'default' : 'outline'}
className="h-7 text-xs px-2"
onClick={() => {
setTtsEngine('edge');
setTtsVoice('fr-FR-VivienneMultilingualNeural');
}}
>
Edge
</Button>
<Button
size="sm"
variant={ttsEngine === 'gemini' ? 'default' : 'outline'}
className="h-7 text-xs px-2"
onClick={() => {
setTtsEngine('gemini');
setTtsVoice('fr-FR-Standard-B');
}}
>
Gemini
</Button>
</div>
</div>
{/* Voice Selector */}
<div className="flex items-center gap-2 mt-2">
<Label className="text-xs font-medium shrink-0">Voix:</Label>
<Select value={ttsVoice} onValueChange={setTtsVoice}>
<SelectTrigger className="h-7 text-xs flex-1">
<SelectValue />
</SelectTrigger>
<SelectContent>
{ttsEngine === 'edge' ? (
edgeVoices.map(v => (
<SelectItem key={v.value} value={v.value}>{v.label}</SelectItem>
))
) : (
geminiVoices.map(v => (
<SelectItem key={v.value} value={v.value}>{v.label}</SelectItem>
))
)}
</SelectContent>
</Select>
</div>
<div className="mt-2">
<Button
size="sm"
variant="secondary"
className="w-full h-8 text-xs"
onClick={async (e) => {
e.stopPropagation();
const textToTest = overlayText || "Ceci est un test de voix.";
try {
const response = await apiRequest('POST', '/api/reels/tts-preview', {
text: textToTest,
ttsEngine,
ttsVoice,
});
const data = await response.json();
if (data.success && data.audioBase64) {
const audio = new Audio(`data:audio/mp3;base64,${data.audioBase64}`);
audio.play();
}
} catch (err) {
toast({ title: "Erreur", description: "Impossible de lire la voix", variant: "destructive" });
}
}}
>
<Play className="w-3 h-3 mr-1" /> Tester la voix
</Button>
<VoicePicker
value={voiceSettings}
onChange={setVoiceSettings}
sampleText={overlayText}
compact
/>
</div>
{syncInfo && (
+12 -64
View File
@@ -1,5 +1,7 @@
import { useState, useRef } from "react";
import { useToast } from "@/hooks/use-toast";
import { VoicePicker, type VoiceSettings } from "@/components/reels/voice-picker";
import { DEFAULT_TTS_STYLE, DEFAULT_VOICE } from "@shared/voices";
import { Card, CardContent, CardHeader, CardTitle } from "@/components/ui/card";
import { Button } from "@/components/ui/button";
import { Textarea } from "@/components/ui/textarea";
@@ -33,23 +35,11 @@ export default function MobileRemotionVideoPage() {
const [productInfo, setProductInfo] = useState("");
const [generatedVariants, setGeneratedVariants] = useState<any[]>([]);
const [ttsEnabled, setTtsEnabled] = useState(true);
const [ttsEngine, setTtsEngine] = useState<'edge' | 'gemini'>('edge');
const [ttsVoice, setTtsVoice] = useState('fr-FR-VivienneMultilingualNeural');
// Gemini native TTS voices (Charon = homme, Kore = femme)
const geminiVoices = [
{ label: 'Charon - Voix Homme', value: 'fr-FR-Standard-B' },
{ label: 'Kore - Voix Femme', value: 'fr-FR-Standard-A' },
];
// French Edge TTS voices
const edgeVoices = [
{ label: 'Vivienne (Femme)', value: 'fr-FR-VivienneMultilingualNeural' },
{ label: 'Henri (Homme)', value: 'fr-FR-HenriNeural' },
{ label: 'Denise (Femme)', value: 'fr-FR-DeniseNeural' },
{ label: 'Rémy (Homme)', value: 'fr-FR-RemyMultilingualNeural' },
{ label: 'Jenny (Anglaise, Femme)', value: 'en-US-JennyNeural' },
];
const [voiceSettings, setVoiceSettings] = useState<VoiceSettings>({
engine: 'gemini',
voice: DEFAULT_VOICE.gemini,
style: DEFAULT_TTS_STYLE,
});
const [musicFile, setMusicFile] = useState<File | null>(null);
const [selectedTrack, setSelectedTrack] = useState<AudioTrack | null>(null);
const [musicVolume, setMusicVolume] = useState(0.3);
@@ -143,16 +133,6 @@ export default function MobileRemotionVideoPage() {
const totalSelected = images.length + selectedLibraryImages.length;
const handleTtsPreview = async () => {
const ttsText = stripForTTS(overlayText);
if (!ttsText) return;
try {
const r = await apiRequest('POST', '/api/reels/tts-preview', { text: ttsText, ttsEngine, ttsVoice });
const data = await r.json();
if (data.success && data.audioBase64) new window.Audio(`data:audio/mp3;base64,${data.audioBase64}`).play();
} catch { toast({ title: "Erreur prévisualisation voix", variant: "destructive" }); }
};
const togglePlayPreview = (track: AudioTrack) => {
if (!audioRef.current) return;
if (isPlaying === track.id) { audioRef.current.pause(); setIsPlaying(null); }
@@ -168,8 +148,10 @@ export default function MobileRemotionVideoPage() {
images.forEach(img => formData.append("images", img));
selectedLibraryImages.forEach(m => formData.append("existingImageUrls", m.originalUrl));
if (overlayText) formData.append("overlayText", overlayText);
formData.append("ttsEngine", ttsEngine);
formData.append("ttsVoice", ttsVoice);
formData.append("ttsEnabled", String(ttsEnabled));
formData.append("ttsEngine", voiceSettings.engine);
formData.append("ttsVoice", voiceSettings.voice);
formData.append("ttsStyle", voiceSettings.style);
if (selectedPageIds[0]) formData.append("selectedPageId", selectedPageIds[0]);
if (musicFile) { formData.append("music", musicFile); formData.append("musicVolume", String(musicVolume)); }
else if (selectedTrack) { formData.append("musicTrackUrl", selectedTrack.url); formData.append("musicVolume", String(musicVolume)); }
@@ -354,41 +336,7 @@ export default function MobileRemotionVideoPage() {
<div className="space-y-2">
<p className="text-xs text-muted-foreground">TTS — voix activée</p>
{/* TTS Engine Selector */}
<div className="flex items-center gap-2">
<Label className="text-xs font-medium">Moteur:</Label>
<div className="flex gap-1">
<Button size="sm" variant={ttsEngine === 'edge' ? 'default' : 'outline'} className="h-7 text-xs px-2"
onClick={() => { setTtsEngine('edge'); setTtsVoice('fr-FR-VivienneMultilingualNeural'); }}>
Edge
</Button>
<Button size="sm" variant={ttsEngine === 'gemini' ? 'default' : 'outline'} className="h-7 text-xs px-2"
onClick={() => { setTtsEngine('gemini'); setTtsVoice('fr-FR-Standard-B'); }}>
Gemini
</Button>
</div>
</div>
{/* Voice Selector */}
<div className="flex items-center gap-2">
<Label className="text-xs font-medium shrink-0">Voix:</Label>
<Select value={ttsVoice} onValueChange={setTtsVoice}>
<SelectTrigger className="h-7 text-xs flex-1">
<SelectValue />
</SelectTrigger>
<SelectContent>
{ttsEngine === 'edge' ? (
edgeVoices.map(v => <SelectItem key={v.value} value={v.value}>{v.label}</SelectItem>)
) : (
geminiVoices.map(v => <SelectItem key={v.value} value={v.value}>{v.label}</SelectItem>)
)}
</SelectContent>
</Select>
</div>
<Button size="sm" variant="outline" className="w-full" onClick={handleTtsPreview} disabled={!overlayText}>
<Volume2 className="mr-2 w-3 h-3" /> Écouter la voix
</Button>
<VoicePicker value={voiceSettings} onChange={setVoiceSettings} sampleText={overlayText} compact />
</div>
)}
</CardContent>
+18 -101
View File
@@ -25,6 +25,8 @@ import {
SelectValue,
} from "@/components/ui/select";
import { useToast } from "@/hooks/use-toast";
import { VoicePicker, type VoiceSettings } from "@/components/reels/voice-picker";
import { DEFAULT_TTS_STYLE, DEFAULT_VOICE } from "@shared/voices";
import { apiRequest, queryClient, handleUnauthorized, getErrorMessage } from "@/lib/queryClient";
import type { SocialPage, Media } from "@shared/schema";
import { SiFacebook, SiTiktok } from "react-icons/si";
@@ -67,7 +69,8 @@ export default function NewReel() {
const [musicVolume, setMusicVolume] = useState([25]);
const [ttsEnabled, setTtsEnabled] = useState(true);
const [drawText, setDrawText] = useState(true);
const [stabilize, setStabilize] = useState(true); // default to true
// Désactivée par défaut : double le temps de rendu, utile seulement pour une vidéo tremblée
const [stabilize, setStabilize] = useState(false);
const [enableEndingEffect, setEnableEndingEffect] = useState(true);
// TTS Sync state
@@ -79,24 +82,13 @@ export default function NewReel() {
warnings: string[];
} | null>(null);
// TTS Engine & Voice
const [ttsEngine, setTtsEngine] = useState<'edge' | 'gemini'>('edge');
const [ttsVoice, setTtsVoice] = useState('fr-FR-VivienneMultilingualNeural');
// Gemini native TTS voices (Charon = homme, Kore = femme)
const geminiVoices = [
{ label: 'Charon - Voix Homme', value: 'fr-FR-Standard-B' },
{ label: 'Kore - Voix Femme', value: 'fr-FR-Standard-A' },
];
// French Edge TTS voices
const edgeVoices = [
{ label: 'Vivienne (Femme)', value: 'fr-FR-VivienneMultilingualNeural' },
{ label: 'Henri (Homme)', value: 'fr-FR-HenriNeural' },
{ label: 'Denise (Femme)', value: 'fr-FR-DeniseNeural' },
{ label: 'Rémy (Homme)', value: 'fr-FR-RemyMultilingualNeural' },
{ label: 'Jenny (Anglaise, Femme)', value: 'en-US-JennyNeural' },
];
// Voix : moteur, voix et ton de lecture
const [voiceSettings, setVoiceSettings] = useState<VoiceSettings>({
engine: 'gemini',
voice: DEFAULT_VOICE.gemini,
style: DEFAULT_TTS_STYLE,
});
const { engine: ttsEngine, voice: ttsVoice, style: ttsStyle } = voiceSettings;
// Enable TTS by default on mobile
useEffect(() => {
@@ -413,6 +405,7 @@ export default function NewReel() {
ttsEnabled,
ttsEngine,
ttsVoice,
ttsStyle,
drawText,
stabilize: stabilize,
enableEndingEffect,
@@ -819,88 +812,12 @@ export default function NewReel() {
<span>TTS — voix activée</span>
</div>
{/* TTS Engine Selector */}
<div className="flex items-center gap-4 mt-3">
<Label className="text-sm font-medium">Moteur:</Label>
<div className="flex gap-2">
<Button
size="sm"
variant={ttsEngine === 'edge' ? 'default' : 'outline'}
onClick={() => {
setTtsEngine('edge');
setTtsVoice('fr-FR-VivienneMultilingualNeural');
}}
>
Edge TTS
</Button>
<Button
size="sm"
variant={ttsEngine === 'gemini' ? 'default' : 'outline'}
onClick={() => {
setTtsEngine('gemini');
setTtsVoice('fr-FR-Standard-B');
}}
>
Gemini TTS
</Button>
</div>
</div>
{/* Voice Selector */}
<div className="flex items-center gap-3 mt-3">
<Label className="text-sm font-medium shrink-0">Voix:</Label>
<Select
value={ttsVoice}
onValueChange={setTtsVoice}
>
<SelectTrigger className="flex-1">
<SelectValue />
</SelectTrigger>
<SelectContent>
{ttsEngine === 'edge' ? (
edgeVoices.map(v => (
<SelectItem key={v.value} value={v.value}>{v.label}</SelectItem>
))
) : (
geminiVoices.map(v => (
<SelectItem key={v.value} value={v.value}>{v.label}</SelectItem>
))
)}
</SelectContent>
</Select>
</div>
<div className="mt-3 flex gap-2">
<Button
size="sm"
variant="secondary"
className="w-full"
onClick={async (e) => {
e.stopPropagation();
const textToTest = overlayText || "Ceci est un test de voix pour votre vidéo.";
try {
const response = await apiRequest('POST', '/api/reels/tts-preview', {
text: textToTest,
ttsEngine,
ttsVoice,
});
const data = await response.json();
if (data.success && data.audioBase64) {
const audio = new Audio(`data:audio/mp3;base64,${data.audioBase64}`);
audio.play();
}
} catch (err) {
toast({
title: "Erreur",
description: "Impossible de tester la voix",
variant: "destructive"
});
}
}}
>
<Play className="w-3 h-3 mr-2" />
Tester la voix
</Button>
<div className="mt-3">
<VoicePicker
value={voiceSettings}
onChange={setVoiceSettings}
sampleText={overlayText}
/>
</div>
{syncInfo && (
+12 -64
View File
@@ -2,6 +2,8 @@ import { useState, useRef } from "react";
import Sidebar from "@/components/sidebar";
import TopBar from "@/components/topbar";
import { useToast } from "@/hooks/use-toast";
import { VoicePicker, type VoiceSettings } from "@/components/reels/voice-picker";
import { DEFAULT_TTS_STYLE, DEFAULT_VOICE } from "@shared/voices";
import { Card, CardContent, CardDescription, CardHeader, CardTitle } from "@/components/ui/card";
import { Button } from "@/components/ui/button";
import { Textarea } from "@/components/ui/textarea";
@@ -34,23 +36,11 @@ export default function RemotionVideoPage() {
const [productInfo, setProductInfo] = useState("");
const [generatedVariants, setGeneratedVariants] = useState<any[]>([]);
const [ttsEnabled, setTtsEnabled] = useState(true);
const [ttsEngine, setTtsEngine] = useState<'edge' | 'gemini'>('edge');
const [ttsVoice, setTtsVoice] = useState('fr-FR-VivienneMultilingualNeural');
// Gemini native TTS voices (Charon = homme, Kore = femme)
const geminiVoices = [
{ label: 'Charon - Voix Homme', value: 'fr-FR-Standard-B' },
{ label: 'Kore - Voix Femme', value: 'fr-FR-Standard-A' },
];
// French Edge TTS voices
const edgeVoices = [
{ label: 'Vivienne (Femme)', value: 'fr-FR-VivienneMultilingualNeural' },
{ label: 'Henri (Homme)', value: 'fr-FR-HenriNeural' },
{ label: 'Denise (Femme)', value: 'fr-FR-DeniseNeural' },
{ label: 'Rémy (Homme)', value: 'fr-FR-RemyMultilingualNeural' },
{ label: 'Jenny (Anglaise, Femme)', value: 'en-US-JennyNeural' },
];
const [voiceSettings, setVoiceSettings] = useState<VoiceSettings>({
engine: 'gemini',
voice: DEFAULT_VOICE.gemini,
style: DEFAULT_TTS_STYLE,
});
const [musicFile, setMusicFile] = useState<File | null>(null);
const [selectedTrack, setSelectedTrack] = useState<AudioTrack | null>(null);
const [musicVolume, setMusicVolume] = useState(0.3);
@@ -109,16 +99,6 @@ export default function RemotionVideoPage() {
});
};
const handleTtsPreview = async () => {
const ttsText = overlayText.replace(/#\w+/g, '').replace(/[\uD800-\uDFFF\u2600-\u27BF]/g, '').replace(/\s+/g, ' ').trim();
if (!ttsText) return;
try {
const r = await apiRequest('POST', '/api/reels/tts-preview', { text: ttsText, ttsEngine, ttsVoice });
const data = await r.json();
if (data.success && data.audioBase64) new window.Audio(`data:audio/mp3;base64,${data.audioBase64}`).play();
} catch { toast({ title: "Erreur prévisualisation voix", variant: "destructive" }); }
};
const togglePlayPreview = (track: AudioTrack) => {
if (!audioRef.current) return;
if (isPlaying === track.id) { audioRef.current.pause(); setIsPlaying(null); }
@@ -134,8 +114,10 @@ export default function RemotionVideoPage() {
images.forEach(img => formData.append("images", img));
selectedLibraryImages.forEach(m => formData.append("existingImageUrls", m.originalUrl));
if (overlayText) formData.append("overlayText", overlayText);
formData.append("ttsEngine", ttsEngine);
formData.append("ttsVoice", ttsVoice);
formData.append("ttsEnabled", String(ttsEnabled));
formData.append("ttsEngine", voiceSettings.engine);
formData.append("ttsVoice", voiceSettings.voice);
formData.append("ttsStyle", voiceSettings.style);
if (selectedPageIds[0]) formData.append("selectedPageId", selectedPageIds[0]);
if (musicFile) { formData.append("music", musicFile); formData.append("musicVolume", String(musicVolume)); }
else if (selectedTrack) { formData.append("musicTrackUrl", selectedTrack.url); formData.append("musicVolume", String(musicVolume)); }
@@ -294,41 +276,7 @@ export default function RemotionVideoPage() {
<div className="space-y-3 p-3 bg-muted/30 rounded-lg border">
<p className="text-sm text-muted-foreground">TTS — voix activée</p>
{/* TTS Engine Selector */}
<div className="flex items-center gap-2">
<Label className="text-xs font-medium">Moteur:</Label>
<div className="flex gap-1">
<Button size="sm" variant={ttsEngine === 'edge' ? 'default' : 'outline'} className="h-7 text-xs"
onClick={() => { setTtsEngine('edge'); setTtsVoice('fr-FR-VivienneMultilingualNeural'); }}>
Edge
</Button>
<Button size="sm" variant={ttsEngine === 'gemini' ? 'default' : 'outline'} className="h-7 text-xs"
onClick={() => { setTtsEngine('gemini'); setTtsVoice('fr-FR-Standard-B'); }}>
Gemini
</Button>
</div>
</div>
{/* Voice Selector */}
<div className="flex items-center gap-2">
<Label className="text-xs font-medium shrink-0">Voix:</Label>
<Select value={ttsVoice} onValueChange={setTtsVoice}>
<SelectTrigger className="h-7 text-xs flex-1">
<SelectValue />
</SelectTrigger>
<SelectContent>
{ttsEngine === 'edge' ? (
edgeVoices.map(v => <SelectItem key={v.value} value={v.value}>{v.label}</SelectItem>)
) : (
geminiVoices.map(v => <SelectItem key={v.value} value={v.value}>{v.label}</SelectItem>)
)}
</SelectContent>
</Select>
</div>
<Button size="sm" variant="outline" className="w-full" onClick={handleTtsPreview} disabled={!overlayText}>
<Volume2 className="mr-2 w-3 h-3" /> Écouter la voix
</Button>
<VoicePicker value={voiceSettings} onChange={setVoiceSettings} sampleText={overlayText} />
</div>
)}
</CardContent>
+4
View File
@@ -71,12 +71,16 @@ services:
build:
context: ./ffmpeg-service
dockerfile: Dockerfile
args:
# Modèle Whisper (calage des sous-titres) intégré à l'image
WHISPER_MODEL: ${WHISPER_MODEL:-base}
image: socialflow-ffmpeg-api:latest
pull_policy: build
container_name: socialflow-ffmpeg
restart: unless-stopped
environment:
API_KEY: ${FFMPEG_API_KEY:-socialflow-secret-ffmpeg-key}
GEMINI_TTS_MODEL: ${GEMINI_TTS_MODEL:-gemini-2.5-flash-preview-tts}
networks:
- internal
expose:
+4
View File
@@ -0,0 +1,4 @@
__pycache__
.pytest_cache
.ruff_cache
tests
+3
View File
@@ -0,0 +1,3 @@
__pycache__/
.pytest_cache/
.ruff_cache/
+23 -22
View File
@@ -1,38 +1,39 @@
FROM python:3.11-slim
# Install system dependencies including FFmpeg with vidstab support and fonts
# We need to build/install libvidstab and compile ffmpeg with it, OR use static build
RUN apt-get update && apt-get install -y \
wget \
xz-utils \
fonts-dejavu \
fontconfig \
build-essential \
python3-dev \
# Polices des sous-titres (Montserrat) et des emojis, installées au build
# plutôt que téléchargées à chaque démarrage.
RUN apt-get update && apt-get install -y --no-install-recommends \
wget xz-utils ca-certificates fontconfig \
fonts-montserrat fonts-dejavu-core fonts-noto-color-emoji \
&& rm -rf /var/lib/apt/lists/*
# Download static FFmpeg build with all filters including vidstab
RUN wget -q https://johnvansickle.com/ffmpeg/releases/ffmpeg-release-amd64-static.tar.xz \
&& tar xf ffmpeg-release-amd64-static.tar.xz \
&& mv ffmpeg-*-amd64-static/ffmpeg /usr/local/bin/ \
&& mv ffmpeg-*-amd64-static/ffprobe /usr/local/bin/ \
# FFmpeg statique (vidstab, zimg pour le tone-mapping HDR, libass).
# Version épinglée : une mise à jour ne change plus le rendu à notre insu.
ARG FFMPEG_VERSION=7.0.2
RUN wget -q https://johnvansickle.com/ffmpeg/releases/ffmpeg-${FFMPEG_VERSION}-amd64-static.tar.xz \
|| wget -q https://johnvansickle.com/ffmpeg/old-releases/ffmpeg-${FFMPEG_VERSION}-amd64-static.tar.xz \
&& tar xf ffmpeg-${FFMPEG_VERSION}-amd64-static.tar.xz \
&& mv ffmpeg-*-amd64-static/ffmpeg ffmpeg-*-amd64-static/ffprobe /usr/local/bin/ \
&& rm -rf ffmpeg-* \
&& fc-cache -fv
&& fc-cache -f
WORKDIR /app
# Copy requirements first for better caching
COPY requirements.txt .
RUN pip install --no-cache-dir -r requirements.txt
# Copy application code
# Modèle Whisper (calage des sous-titres) téléchargé au build : aucun accès
# réseau nécessaire au premier rendu.
ARG WHISPER_MODEL=base
ENV WHISPER_MODEL=${WHISPER_MODEL} \
HF_HOME=/opt/hf-cache \
HOME=/tmp \
XDG_CACHE_HOME=/tmp/.cache
RUN python -c "from faster_whisper import WhisperModel; WhisperModel('${WHISPER_MODEL}', device='cpu', compute_type='int8')"
COPY main.py .
COPY app ./app
# Create API Key env var (should be overridden in docker-compose)
ENV API_KEY=default-dev-key
# Expose port
EXPOSE 8000
# Run the application
CMD ["uvicorn", "main:app", "--host", "0.0.0.0", "--port", "8000"]
Binary file not shown.
+1
View File
@@ -0,0 +1 @@
"""Service audio et vidéo des Reels SocialFlow."""
+132
View File
@@ -0,0 +1,132 @@
"""Calage mot à mot du texte affiché sur la voix.
Le texte affiché (ponctuation, majuscules) diffère légèrement des mots
reconnus ou annoncés par le moteur de voix : on apparie les deux séquences
(difflib) et on interpole les mots sans correspondance. Remplace ffsubsync,
qui ne faisait qu'un décalage global du texte réparti uniformément.
"""
import asyncio
import difflib
import logging
import re
import unicodedata
from dataclasses import dataclass
from functools import lru_cache
from pathlib import Path
from . import config
log = logging.getLogger(__name__)
@dataclass
class Word:
text: str
start: float
end: float
def to_dict(self) -> dict:
return {"text": self.text, "start": round(self.start, 3), "end": round(self.end, 3)}
def normalize(token: str) -> str:
"""Forme comparable d'un mot : minuscules, sans accents ni ponctuation."""
decomposed = unicodedata.normalize("NFKD", token.lower())
stripped = "".join(c for c in decomposed if not unicodedata.combining(c))
return re.sub(r"[^a-z0-9]", "", stripped)
def display_tokens(text: str) -> list[str]:
"""Mots à afficher ; la ponctuation isolée est rattachée au mot précédent."""
tokens: list[str] = []
for raw in text.split():
if tokens and not normalize(raw):
tokens[-1] += raw if raw in ",.!?;:…" else f" {raw}"
else:
tokens.append(raw)
return tokens
def align_words(display_text: str, spoken: list[Word], total_duration: float | None = None) -> list[Word]:
"""Attribue à chaque mot affiché un début et une fin tirés des mots prononcés."""
tokens = display_tokens(display_text)
if not tokens:
return []
if not spoken:
return _spread(tokens, 0.0, total_duration or len(tokens) * 0.4)
timed: list[Word | None] = [None] * len(tokens)
matcher = difflib.SequenceMatcher(
a=[normalize(t) for t in tokens], b=[normalize(w.text) for w in spoken], autojunk=False
)
for block in matcher.get_matching_blocks():
for k in range(block.size):
src = spoken[block.b + k]
timed[block.a + k] = Word(tokens[block.a + k], src.start, src.end)
# Mots non appariés : répartis dans le trou entre leurs voisins datés
end_of_speech = max(spoken[-1].end, total_duration or 0.0)
i = 0
while i < len(tokens):
if timed[i] is not None:
i += 1
continue
j = i
while j < len(tokens) and timed[j] is None:
j += 1
gap_start = timed[i - 1].end if i > 0 else 0.0
gap_end = timed[j].start if j < len(tokens) else end_of_speech
if gap_end - gap_start < 0.05 * (j - i):
gap_end = gap_start + 0.25 * (j - i)
timed[i:j] = _spread(tokens[i:j], gap_start, gap_end)
i = j
words = [w for w in timed if w is not None]
# Monotonie stricte : un mot ne commence jamais avant la fin du précédent
for prev, cur in zip(words, words[1:]):
cur.start = max(cur.start, prev.start + 0.01)
cur.end = max(cur.end, cur.start + 0.05)
return words
def _spread(tokens: list[str], start: float, end: float) -> list[Word]:
"""Répartit des mots sur un intervalle au prorata de leur longueur."""
weights = [max(1, len(normalize(t))) for t in tokens]
total = sum(weights)
words, cursor = [], start
for token, weight in zip(tokens, weights):
duration = (end - start) * weight / total
words.append(Word(token, cursor, cursor + duration))
cursor += duration
return words
@lru_cache(maxsize=1)
def _model():
from faster_whisper import WhisperModel
log.info("Chargement du modèle Whisper %s", config.WHISPER_MODEL)
return WhisperModel(config.WHISPER_MODEL, device="cpu", compute_type=config.WHISPER_COMPUTE_TYPE)
def _transcribe_sync(audio: Path, hint: str | None) -> list[Word]:
segments, _ = _model().transcribe(
str(audio),
language="fr",
word_timestamps=True,
initial_prompt=hint or None,
vad_filter=False,
beam_size=5,
)
return [
Word(w.word.strip(), float(w.start), float(w.end))
for segment in segments
for w in (segment.words or [])
if w.word.strip()
]
async def transcribe(audio: Path, hint: str | None = None) -> list[Word]:
"""Mots prononcés et leurs instants (Whisper, exécuté hors de la boucle async)."""
return await asyncio.to_thread(_transcribe_sync, audio, hint)
+304
View File
@@ -0,0 +1,304 @@
"""API HTTP du service : voix, rendu des Reels et récupération des fichiers."""
import asyncio
import base64
import contextlib
import logging
import time
from pathlib import Path
import httpx
from fastapi import Depends, FastAPI, Header, HTTPException
from fastapi.responses import FileResponse
from pydantic import BaseModel
from . import align, config, jobs, proc, render, subtitles, tts, voices
from .audio import encode_preview
from .text import clean_text
logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s: %(message)s")
log = logging.getLogger("reels")
if not config.API_KEY:
raise RuntimeError("API_KEY doit être défini : le service refuse de démarrer sans clé.")
# Un seul encodage à la fois : ils saturent le CPU, les suivants attendent
render_slot = asyncio.Semaphore(1)
@contextlib.asynccontextmanager
async def lifespan(_app: FastAPI):
removed = jobs.purge_expired()
log.info("Service prêt (%d dossier(s) expiré(s) purgé(s))", removed)
async def purge_loop():
while True:
await asyncio.sleep(600)
jobs.purge_expired()
task = asyncio.create_task(purge_loop())
yield
task.cancel()
app = FastAPI(title="SocialFlow Reels", lifespan=lifespan)
def require_key(x_api_key: str | None = Header(None)) -> None:
if x_api_key != config.API_KEY:
raise HTTPException(status_code=401, detail="Invalid API Key")
class TtsRequest(BaseModel):
text: str
tts_voice: str | None = None
tts_engine: str | None = "gemini"
tts_style: str | None = None
gemini_api_key: str | None = None
class ReelRequest(BaseModel):
video_base64: str | None = None
video_url: str | None = None
text: str | None = None
music_url: str | None = None
watermark_url: str | None = None
store_name: str | None = None
font_size: int = 64
music_volume: float = 0.25
tts_enabled: bool = False
tts_voice: str | None = None
tts_engine: str | None = "gemini"
tts_style: str | None = None
gemini_api_key: str | None = None
draw_text: bool = True
stabilize: bool = False
enable_ending_effect: bool = True
# Champs d'anciennes versions, acceptés et ignorés
music_id: str | None = None
word_duration: float | None = None
@app.get("/health", dependencies=[Depends(require_key)])
async def health():
return {"status": "ok", "version": 2}
@app.get("/voices", dependencies=[Depends(require_key)])
async def list_voices():
return voices.catalog()
@app.post("/preview-tts", dependencies=[Depends(require_key)])
async def preview_tts(request: TtsRequest):
text = clean_text(request.text)
if not text:
raise HTTPException(status_code=400, detail="Texte vide après nettoyage (emojis et hashtags retirés)")
job_id, workdir = jobs.new_job()
try:
track = await _synthesize(text, request.text, request, workdir)
preview = workdir / "preview.mp3"
await encode_preview(track.path, preview)
return {
"success": True,
"audio_base64": base64.b64encode(preview.read_bytes()).decode(),
"duration": track.duration,
"words": [w.to_dict() for w in track.words],
"engine": track.engine,
"voice": track.voice,
"warnings": track.warnings,
}
finally:
jobs.remove_job(job_id)
@app.post("/process-reel", dependencies=[Depends(require_key)])
async def process_reel(request: ReelRequest):
"""Produit le Reel ; le MP4 se récupère ensuite via GET /files/{job_id}/output.mp4."""
job_id, workdir = jobs.new_job()
stats: dict[str, float] = {}
started = time.monotonic()
try:
async with render_slot:
return await _process(request, job_id, workdir, stats, started)
except HTTPException:
jobs.remove_job(job_id)
raise
except proc.CommandError as error:
jobs.remove_job(job_id)
log.error("Rendu %s en échec : %s", job_id, error)
raise HTTPException(status_code=500, detail=str(error)) from error
except Exception as error:
jobs.remove_job(job_id)
log.exception("Rendu %s en échec", job_id)
raise HTTPException(status_code=500, detail=f"Erreur de rendu : {error}") from error
@app.get("/files/{job_id}/{name}", dependencies=[Depends(require_key)])
async def get_file(job_id: str, name: str):
path = jobs.job_file(job_id, name)
if not path:
raise HTTPException(status_code=404, detail="Fichier introuvable ou expiré")
return FileResponse(path, media_type="video/mp4" if name.endswith(".mp4") else None)
@app.delete("/jobs/{job_id}", dependencies=[Depends(require_key)])
async def delete_job(job_id: str):
jobs.remove_job(job_id)
return {"success": True}
# --- Orchestration --------------------------------------------------------
async def _synthesize(text: str, display_source: str | None, request, workdir: Path) -> tts.VoiceTrack:
try:
return await tts.synthesize(
text=text,
display_text=clean_text(display_source) or text,
engine=request.tts_engine,
voice=request.tts_voice,
style=request.tts_style,
gemini_api_key=request.gemini_api_key,
workdir=workdir,
)
except Exception as error:
log.exception("Voix impossible à générer")
raise HTTPException(status_code=502, detail=f"La voix n'a pas pu être générée : {error}") from error
async def _download(url: str, target: Path, what: str, required: bool) -> bool:
try:
async with httpx.AsyncClient(timeout=httpx.Timeout(120, connect=15), follow_redirects=True) as client:
async with client.stream("GET", url, headers={"User-Agent": "Mozilla/5.0"}) as response:
response.raise_for_status()
with target.open("wb") as out:
async for chunk in response.aiter_bytes(1 << 20):
out.write(chunk)
return True
except Exception as error:
if required:
detail = f"Téléchargement de {what} impossible : {error}"
raise HTTPException(status_code=400, detail=detail) from error
log.warning("Téléchargement de %s impossible, on continue sans : %s", what, error)
return False
async def _process(request: ReelRequest, job_id: str, workdir: Path, stats: dict, started: float) -> dict:
step = time.monotonic()
video = workdir / "input.mp4"
if request.video_base64:
video.write_bytes(base64.b64decode(request.video_base64))
elif request.video_url:
await _download(request.video_url, video, "la vidéo", required=True)
else:
raise HTTPException(status_code=400, detail="Aucune vidéo fournie")
music = workdir / "music.audio"
has_music = bool(request.music_url) and await _download(
request.music_url, music, "la musique", required=False
)
watermark = workdir / "watermark.png"
has_watermark = bool(request.watermark_url) and await _download(
request.watermark_url, watermark, "le logo", required=False
)
info = await proc.probe(video)
if info.duration <= 0:
raise HTTPException(status_code=400, detail="Vidéo illisible (durée nulle)")
stats["download"] = time.monotonic() - step
# --- Voix ---
step = time.monotonic()
track = None
spoken_text = clean_text(request.text)
if request.tts_enabled and spoken_text:
track = await _synthesize(spoken_text, request.text, request, workdir)
stats["tts"] = time.monotonic() - step
plan = render.RenderPlan(
video=video,
video_duration=info.duration,
output=workdir / "output.mp4",
is_hdr=info.is_hdr,
music=music if has_music else None,
music_volume=request.music_volume,
voice=track.path if track else None,
voice_duration=track.duration if track else 0.0,
watermark=watermark if has_watermark else None,
ending_effect=request.enable_ending_effect,
keep_original_audio=info.has_audio,
)
# --- Sous-titres ---
step = time.monotonic()
font_size = max(48, round(request.font_size * 1.4))
display = clean_text(request.text)
if request.draw_text and display:
captions = workdir / "captions.ass"
if track:
subtitles.write_captions(track.words, captions, offset=plan.voice_delay, font_size=font_size)
else:
words = await _caption_words_without_voice(display, video, info, plan)
subtitles.write_captions(words, captions, offset=0.0, font_size=font_size)
plan.captions = captions
if request.store_name and request.enable_ending_effect and has_watermark:
outro = workdir / "outro.ass"
subtitles.write_outro(request.store_name, outro, plan.logo_start, plan.total_duration)
plan.outro = outro
stats["subtitles"] = time.monotonic() - step
# --- Stabilisation (1re passe) ---
step = time.monotonic()
if request.stabilize:
transforms = workdir / "transforms.trf"
try:
await proc.run(render.stabilize_detect_command(video, transforms), timeout=600)
plan.stabilize_transforms = transforms
except proc.CommandError as error:
log.warning("Stabilisation ignorée : %s", error)
stats["stabilize"] = time.monotonic() - step
# --- Encodage ---
step = time.monotonic()
await proc.run(render.build_command(plan), timeout=1200)
stats["encode"] = time.monotonic() - step
stats["total"] = time.monotonic() - started
duration = (await proc.probe(plan.output)).duration
log.info(
"Rendu %s terminé : %.1f s de vidéo, étapes %s",
job_id,
duration,
{k: round(v, 1) for k, v in stats.items()},
)
# Seul le résultat est conservé jusqu'au téléchargement
for entry in workdir.iterdir():
if entry != plan.output:
entry.unlink(missing_ok=True)
return {
"success": True,
"job_id": job_id,
"output_path": f"/files/{job_id}/output.mp4",
"duration": duration,
"tts_engine": track.engine if track else None,
"tts_voice": track.voice if track else None,
"warnings": track.warnings if track else [],
"processing_stats": stats,
}
async def _caption_words_without_voice(display: str, video: Path, info, plan: render.RenderPlan):
"""Sans voix de synthèse : calage sur la parole de la vidéo si elle en contient,
sinon texte réparti sur la durée (hors effet de fin)."""
if info.has_audio:
try:
spoken = await align.transcribe(video, hint=display)
if len(spoken) >= max(2, len(display.split()) // 3):
return align.align_words(display, spoken, info.duration)
except Exception as error: # noqa: BLE001 — on retombe sur la répartition
log.warning("Transcription de la vidéo impossible : %s", error)
end = plan.logo_start if plan.ending_effect and plan.watermark else plan.total_duration
return align.align_words(display, [], max(1.0, end - 0.5))
+60
View File
@@ -0,0 +1,60 @@
"""Traitement de la voix : filtrage, compression et niveau sonore constant."""
from pathlib import Path
from . import proc
# Coupe les basses inutiles, lisse la dynamique et amène la voix à -16 LUFS :
# elle reste intelligible par-dessus la musique sans saturer.
VOICE_CHAIN = (
"highpass=f=80,"
"acompressor=threshold=0.1:ratio=3:attack=5:release=120:makeup=2,"
"loudnorm=I=-16:TP=-1.5:LRA=7,"
"aresample=48000"
)
async def process_voice(source: Path, target: Path) -> None:
"""Produit un WAV 48 kHz mono prêt à mixer."""
await proc.run(
[
"ffmpeg",
"-y",
"-hide_banner",
"-loglevel",
"error",
"-i",
str(source),
"-af",
VOICE_CHAIN,
"-ac",
"1",
"-ar",
"48000",
"-c:a",
"pcm_s16le",
str(target),
],
timeout=120,
)
async def encode_preview(source: Path, target: Path) -> None:
"""MP3 de bonne qualité pour l'écoute dans le navigateur."""
await proc.run(
[
"ffmpeg",
"-y",
"-hide_banner",
"-loglevel",
"error",
"-i",
str(source),
"-c:a",
"libmp3lame",
"-b:a",
"160k",
str(target),
],
timeout=120,
)
+32
View File
@@ -0,0 +1,32 @@
"""Configuration lue dans l'environnement."""
import os
from pathlib import Path
# Pas de clé par défaut : un service exposé avec une clé connue de tous
# n'est pas protégé. Le démarrage échoue si elle manque.
API_KEY = os.environ.get("API_KEY", "")
TEMP_DIR = Path(os.environ.get("TEMP_DIR", "/tmp/ffmpeg_processing"))
# Durée de conservation des fichiers produits (téléchargés par l'application
# puis supprimés ; la purge ne rattrape que les oublis).
FILE_TTL_SECONDS = int(os.environ.get("FILE_TTL_SECONDS", "3600"))
GEMINI_TTS_MODEL = os.environ.get("GEMINI_TTS_MODEL", "gemini-2.5-flash-preview-tts")
# Modèle Whisper utilisé pour caler les sous-titres sur la voix.
WHISPER_MODEL = os.environ.get("WHISPER_MODEL", "base")
WHISPER_COMPUTE_TYPE = os.environ.get("WHISPER_COMPUTE_TYPE", "int8")
# Proxy sortant éventuel pour Edge TTS (aiohttp ne lit pas HTTPS_PROXY seul).
OUTBOUND_PROXY = os.environ.get("HTTPS_PROXY") or os.environ.get("https_proxy") or None
# Rendu
WIDTH = 1080
HEIGHT = 1920
FPS = 30
# La voix démarre après ce délai (le temps de capter l'attention).
VOICE_DELAY = 2.0
SUBTITLE_FONT = os.environ.get("SUBTITLE_FONT", "Montserrat")
+44
View File
@@ -0,0 +1,44 @@
"""Dossiers de travail des rendus et purge des fichiers oubliés."""
import re
import shutil
import time
import uuid
from pathlib import Path
from . import config
_SAFE_ID = re.compile(r"^[0-9a-f-]{36}$")
_SAFE_NAME = re.compile(r"^[\w.-]+$")
def new_job() -> tuple[str, Path]:
job_id = str(uuid.uuid4())
workdir = config.TEMP_DIR / job_id
workdir.mkdir(parents=True)
return job_id, workdir
def job_file(job_id: str, name: str) -> Path | None:
"""Fichier d'un job, ou None si l'identifiant ou le nom sont suspects."""
if not _SAFE_ID.match(job_id) or not _SAFE_NAME.match(name):
return None
path = config.TEMP_DIR / job_id / name
return path if path.is_file() else None
def remove_job(job_id: str) -> None:
if _SAFE_ID.match(job_id):
shutil.rmtree(config.TEMP_DIR / job_id, ignore_errors=True)
def purge_expired() -> int:
"""Supprime les dossiers plus vieux que FILE_TTL_SECONDS."""
config.TEMP_DIR.mkdir(parents=True, exist_ok=True)
limit = time.time() - config.FILE_TTL_SECONDS
removed = 0
for entry in config.TEMP_DIR.iterdir():
if entry.is_dir() and entry.stat().st_mtime < limit:
shutil.rmtree(entry, ignore_errors=True)
removed += 1
return removed
+78
View File
@@ -0,0 +1,78 @@
"""Exécution asynchrone de FFmpeg/ffprobe : le serveur reste disponible
pendant un encodage (l'ancien subprocess.run bloquait toute l'API)."""
import asyncio
import json
import logging
from dataclasses import dataclass
from pathlib import Path
log = logging.getLogger(__name__)
class CommandError(RuntimeError):
"""Commande externe en échec, avec la fin de sa sortie d'erreur."""
def __init__(self, cmd: list[str], returncode: int, stderr: str):
tail = "\n".join(stderr.strip().splitlines()[-15:])
super().__init__(f"{cmd[0]} a échoué (code {returncode}) :\n{tail}")
self.returncode = returncode
self.stderr = stderr
async def run(cmd: list[str], timeout: float = 900) -> str:
"""Lance une commande et renvoie sa sortie standard."""
log.debug("exec: %s", " ".join(cmd))
process = await asyncio.create_subprocess_exec(
*cmd, stdout=asyncio.subprocess.PIPE, stderr=asyncio.subprocess.PIPE
)
try:
stdout, stderr = await asyncio.wait_for(process.communicate(), timeout)
except TimeoutError as error:
process.kill()
await process.wait()
raise CommandError(cmd, -1, f"délai de {timeout:.0f} s dépassé") from error
if process.returncode != 0:
raise CommandError(cmd, process.returncode, stderr.decode(errors="replace"))
return stdout.decode(errors="replace")
@dataclass
class MediaInfo:
duration: float
has_audio: bool
width: int = 0
height: int = 0
color_transfer: str = ""
@property
def is_hdr(self) -> bool:
# HLG (iPhone) ou PQ (HDR10)
return self.color_transfer in ("arib-std-b67", "smpte2084")
async def probe(path: Path) -> MediaInfo:
out = await run(
[
"ffprobe",
"-v",
"error",
"-print_format",
"json",
"-show_format",
"-show_streams",
str(path),
],
timeout=60,
)
data = json.loads(out)
streams = data.get("streams", [])
video = next((s for s in streams if s.get("codec_type") == "video"), {})
duration = float(data.get("format", {}).get("duration") or video.get("duration") or 0)
return MediaInfo(
duration=duration,
has_audio=any(s.get("codec_type") == "audio" for s in streams),
width=int(video.get("width") or 0),
height=int(video.get("height") or 0),
color_transfer=video.get("color_transfer") or "",
)
+217
View File
@@ -0,0 +1,217 @@
"""Construction de la commande FFmpeg d'un Reel (fonction pure, testable)."""
from dataclasses import dataclass
from pathlib import Path
from . import config
from .subtitles import filter_path
FADE_SECONDS = 2.0
LOGO_SECONDS = 5.0
# Silence laissé après la dernière phrase avant la fin de la vidéo
VOICE_TAIL = 0.8
# Durée minimale de l'effet de fin (grand logo + nom du magasin) après la voix
OUTRO_MIN = 2.5
# HDR (HLG/PQ) → SDR BT.709 : sans cela, les vidéos iPhone sortent ternes
TONEMAP = (
"zscale=t=linear:npl=100,format=gbrpf32le,zscale=p=bt709,"
"tonemap=tonemap=hable:desat=0,zscale=t=bt709:m=bt709:r=tv,format=yuv420p"
)
@dataclass
class RenderPlan:
video: Path
video_duration: float
output: Path
is_hdr: bool = False
stabilize_transforms: Path | None = None
music: Path | None = None
music_volume: float = 0.25
voice: Path | None = None
voice_duration: float = 0.0
voice_delay: float = config.VOICE_DELAY
captions: Path | None = None
watermark: Path | None = None
outro: Path | None = None # nom du magasin (effet de fin)
ending_effect: bool = True
keep_original_audio: bool = False
@property
def speech_end(self) -> float:
return self.voice_delay + self.voice_duration if self.voice else 0.0
@property
def total_duration(self) -> float:
"""La vidéo s'allonge (dernière image figée) si la voix dure plus longtemps."""
duration = self.video_duration
if self.voice:
duration = max(duration, self.speech_end + VOICE_TAIL)
if self.ending_effect and self.watermark:
duration = max(duration, self.speech_end + OUTRO_MIN)
return round(duration, 3)
@property
def logo_start(self) -> float:
"""Le grand logo n'arrive qu'une fois la voix terminée."""
return max(0.0, self.total_duration - LOGO_SECONDS, self.speech_end)
@property
def freeze_duration(self) -> float:
return max(0.0, self.total_duration - self.video_duration)
def build_command(plan: RenderPlan) -> list[str]:
total = plan.total_duration
logo_start = plan.logo_start
fade_start = max(0.0, total - FADE_SECONDS)
cmd = ["ffmpeg", "-y", "-hide_banner", "-loglevel", "error", "-i", str(plan.video)]
index = 1
music_idx = voice_idx = wm_idx = None
if plan.music:
# Musique bouclée : une piste plus courte que la vidéo ne coupe plus le son
cmd += ["-stream_loop", "-1", "-i", str(plan.music)]
music_idx, index = index, index + 1
if plan.voice:
cmd += ["-i", str(plan.voice)]
voice_idx, index = index, index + 1
if plan.watermark:
cmd += ["-i", str(plan.watermark)]
wm_idx, index = index, index + 1
graph: list[str] = []
# --- Vidéo ---
chain = []
if plan.stabilize_transforms:
chain.append(
f"vidstabtransform=input={filter_path(plan.stabilize_transforms)}:smoothing=30:relative=1:zoom=5,"
"unsharp=5:5:0.6:5:5:0.0"
)
if plan.is_hdr:
chain.append(TONEMAP)
chain.append(
f"scale={config.WIDTH}:{config.HEIGHT}:force_original_aspect_ratio=increase:flags=lanczos,"
f"crop={config.WIDTH}:{config.HEIGHT},setsar=1,fps={config.FPS}"
)
if plan.freeze_duration > 0:
chain.append(f"tpad=stop_mode=clone:stop_duration={plan.freeze_duration:.3f}")
if plan.captions:
chain.append(f"subtitles='{filter_path(plan.captions)}'")
graph.append(f"[0:v]{','.join(chain)}[vbase]")
current = "vbase"
if wm_idx is not None:
corner = "W-w-30:H-h-30"
if plan.outro and plan.ending_effect:
graph.append(f"[{wm_idx}:v]scale=200:-1,split=2[wm_small][wm_big0]")
graph.append("[wm_big0]scale=-1:300[wm_big]")
graph.append(f"[{current}][wm_small]overlay={corner}:enable='lt(t,{logo_start:.3f})'[vwm1]")
graph.append(f"[vwm1][wm_big]overlay=(W-w)/2:(H-h)/2-100:enable='gte(t,{logo_start:.3f})'[vwm2]")
current = "vwm2"
else:
graph.append(f"[{wm_idx}:v]scale=200:-1[wm_small]")
graph.append(f"[{current}][wm_small]overlay={corner}[vwm1]")
current = "vwm1"
if plan.outro and plan.ending_effect:
graph.append(f"[{current}]subtitles='{filter_path(plan.outro)}'[vout0]")
current = "vout0"
tail = []
if plan.ending_effect:
tail.append(f"fade=t=out:st={fade_start:.3f}:d={FADE_SECONDS}")
tail.append("format=yuv420p")
graph.append(f"[{current}]{','.join(tail)}[vout]")
# --- Audio ---
mix = None
if voice_idx is not None:
delay_ms = int(plan.voice_delay * 1000)
graph.append(f"[{voice_idx}:a]aresample=48000,adelay={delay_ms}:all=1,apad[voice]")
if music_idx is not None:
graph.append(f"[{music_idx}:a]aresample=48000,volume={plan.music_volume:.3f}[music]")
if voice_idx is not None and music_idx is not None:
# La musique baisse automatiquement quand la voix parle (ducking)
graph.append("[voice]asplit=2[vmix][vkey]")
graph.append("[music][vkey]sidechaincompress=threshold=0.02:ratio=8:attack=20:release=400[ducked]")
graph.append("[ducked][vmix]amix=inputs=2:duration=longest:normalize=0[mix]")
mix = "mix"
elif voice_idx is not None:
mix = "voice"
elif music_idx is not None:
mix = "music"
audio_map: list[str] = []
if mix:
# Niveau final recommandé par les réseaux sociaux (~ -14 LUFS)
audio_tail = ["loudnorm=I=-14:TP=-1.5:LRA=11", "aresample=48000"]
if plan.ending_effect:
audio_tail.append(f"afade=t=out:st={fade_start:.3f}:d={FADE_SECONDS}")
graph.append(f"[{mix}]{','.join(audio_tail)}[aout]")
audio_map = ["-map", "[aout]"]
elif plan.keep_original_audio:
audio_map = ["-map", "0:a:0"]
cmd += ["-filter_complex", ";".join(graph), "-map", "[vout]", *audio_map]
cmd += [
"-t",
f"{total:.3f}",
"-c:v",
"libx264",
"-preset",
"medium",
"-crf",
"19",
"-maxrate",
"10M",
"-bufsize",
"20M",
"-profile:v",
"high",
"-level",
"4.2",
"-g",
str(config.FPS * 2),
"-keyint_min",
str(config.FPS),
"-pix_fmt",
"yuv420p",
"-color_primaries",
"bt709",
"-color_trc",
"bt709",
"-colorspace",
"bt709",
"-c:a",
"aac",
"-b:a",
"192k",
"-ar",
"48000",
"-ac",
"2",
"-movflags",
"+faststart",
str(plan.output),
]
return cmd
def stabilize_detect_command(video: Path, transforms: Path) -> list[str]:
return [
"ffmpeg",
"-y",
"-hide_banner",
"-loglevel",
"error",
"-i",
str(video),
"-vf",
f"vidstabdetect=stepsize=32:shakiness=8:accuracy=15:result={filter_path(transforms)}",
"-f",
"null",
"-",
]
+128
View File
@@ -0,0 +1,128 @@
"""Sous-titres ASS : mot à mot, style « Reels » lisible sur tout fond."""
from pathlib import Path
from . import config
from .align import Word
# Couleurs ASS au format &HAABBGGRR
YELLOW = "&H0000E6FF"
WHITE = "&H00FFFFFF"
BLACK = "&H00000000"
SHADOW = "&H64000000"
# Au-delà, une ligne déborde ou se lit mal en un coup d'œil
MAX_WORDS_PER_LINE = 3
MAX_CHARS_PER_LINE = 18
# Bas du texte à ~70 % de la hauteur : au-dessus de la légende et des boutons
# qu'Instagram et TikTok superposent en bas de l'écran.
MARGIN_V = 560
POP_MS = 90
def ass_time(seconds: float) -> str:
seconds = max(0.0, seconds)
centis = int(round(seconds * 100))
hours, centis = divmod(centis, 360000)
minutes, centis = divmod(centis, 6000)
secs, centis = divmod(centis, 100)
return f"{hours}:{minutes:02d}:{secs:02d}.{centis:02d}"
def escape(text: str) -> str:
return text.replace("\\", "").replace("{", "(").replace("}", ")")
def header(styles: list[str]) -> str:
return (
"[Script Info]\n"
"ScriptType: v4.00+\n"
f"PlayResX: {config.WIDTH}\n"
f"PlayResY: {config.HEIGHT}\n"
"ScaledBorderAndShadow: yes\n"
"WrapStyle: 2\n\n"
"[V4+ Styles]\n"
"Format: Name, Fontname, Fontsize, PrimaryColour, SecondaryColour, OutlineColour, "
"BackColour, Bold, Italic, Underline, StrikeOut, ScaleX, ScaleY, Spacing, Angle, "
"BorderStyle, Outline, Shadow, Alignment, MarginL, MarginR, MarginV, Encoding\n"
+ "".join(f"{s}\n" for s in styles)
+ "\n[Events]\n"
"Format: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text\n"
)
def group_lines(words: list[Word]) -> list[list[Word]]:
"""Découpe en lignes courtes, coupées de préférence à la ponctuation."""
lines: list[list[Word]] = []
current: list[Word] = []
for word in words:
chars = sum(len(w.text) + 1 for w in current) + len(word.text)
if current and (len(current) >= MAX_WORDS_PER_LINE or chars > MAX_CHARS_PER_LINE):
lines.append(current)
current = []
current.append(word)
if word.text.endswith((".", "!", "?", ",", ":", ";", "…")):
lines.append(current)
current = []
if current:
lines.append(current)
return lines
def karaoke_line(line: list[Word], line_end: float) -> str:
"""Chaque mot passe du blanc au jaune et « saute » légèrement quand il est dit."""
parts = []
line_start = line[0].start
for index, word in enumerate(line):
next_start = line[index + 1].start if index + 1 < len(line) else line_end
fill_cs = max(1, int(round((next_start - word.start) * 100)))
t0 = int(round((word.start - line_start) * 1000))
pop = (
f"\\fscx100\\fscy100"
f"\\t({t0},{t0 + POP_MS},\\fscx112\\fscy112)"
f"\\t({t0 + POP_MS},{t0 + 2 * POP_MS},\\fscx100\\fscy100)"
)
parts.append(f"{{\\kf{fill_cs}{pop}}}{escape(word.text)}")
return " ".join(parts)
def write_captions(
words: list[Word], path: Path, *, offset: float, font_size: int, end_time: float | None = None
) -> None:
"""Écrit le fichier ASS des sous-titres, décalés de `offset` secondes."""
style = (
f"Style: Caption,{config.SUBTITLE_FONT},{font_size},{YELLOW},{WHITE},{BLACK},{SHADOW},"
f"-1,0,0,0,100,100,0,0,1,6,3,2,80,80,{MARGIN_V},1"
)
lines = group_lines(words)
events = []
for index, line in enumerate(lines):
start = line[0].start
natural_end = line[-1].end + 0.25
if index + 1 < len(lines):
end = min(max(natural_end, line[-1].end), lines[index + 1][0].start)
else:
end = max(natural_end, end_time - offset if end_time else natural_end)
end = max(end, start + 0.3)
events.append(
f"Dialogue: 0,{ass_time(start + offset)},{ass_time(end + offset)},Caption,,0,0,0,,"
f"{karaoke_line(line, end)}"
)
path.write_text(header([style]) + "\n".join(events) + "\n", encoding="utf-8")
def write_outro(store_name: str, path: Path, start: float, end: float) -> None:
"""Nom du magasin sous le logo, en fondu, pendant les dernières secondes."""
style = (
f"Style: Outro,{config.SUBTITLE_FONT},72,{WHITE},{WHITE},{BLACK},{SHADOW},"
"-1,0,0,0,100,100,0,0,1,3,4,2,60,60,700,1"
)
event = (
f"Dialogue: 0,{ass_time(start)},{ass_time(end)},Outro,,0,0,0,,{{\\fad(1200,0)}}{escape(store_name)}"
)
path.write_text(header([style]) + event + "\n", encoding="utf-8")
def filter_path(path: Path) -> str:
"""Chemin utilisable dans un filtre FFmpeg (échappement de : et ')."""
return str(path).replace("\\", "/").replace(":", "\\:").replace("'", "\\'")
+20
View File
@@ -0,0 +1,20 @@
"""Nettoyage du texte avant synthèse et affichage."""
import re
import emoji
_HIDDEN = str.maketrans("", "", "​")
_HASHTAG = re.compile(r"#[\wÀ-ɏ]+")
_URL = re.compile(r"https?://\S+")
def clean_text(text: str | None) -> str:
"""Retire emojis, hashtags, liens et caractères invisibles."""
if not text:
return ""
text = text.translate(_HIDDEN)
text = emoji.replace_emoji(text, replace="")
text = _URL.sub("", text)
text = _HASHTAG.sub("", text)
return " ".join(text.split())
+63
View File
@@ -0,0 +1,63 @@
"""Synthèse vocale : moteur au choix, repli sur Edge, voix traitée et mots minutés."""
import logging
from dataclasses import dataclass, field
from pathlib import Path
from .. import align, audio, proc
from ..align import Word
from . import edge, gemini
log = logging.getLogger(__name__)
@dataclass
class VoiceTrack:
path: Path # WAV 48 kHz traité
duration: float
words: list[Word] # mots affichés, minutés depuis le début de la voix
engine: str
voice: str
warnings: list[str] = field(default_factory=list)
async def synthesize(
*,
text: str,
display_text: str,
engine: str | None,
voice: str | None,
style: str | None,
gemini_api_key: str | None,
workdir: Path,
) -> VoiceTrack:
warnings: list[str] = []
raw: Path | None = None
spoken: list[Word] = []
used_engine, used_voice = "edge", ""
if (engine or "gemini") == "gemini":
if not gemini_api_key:
warnings.append("Clé Gemini absente : voix Edge utilisée à la place.")
else:
try:
raw, used_voice = await gemini.synthesize(text, voice, style, gemini_api_key, workdir)
used_engine = "gemini"
except Exception as error: # noqa: BLE001 — repli sur Edge
log.warning("Gemini TTS en échec, repli sur Edge : %s", error)
warnings.append(f"Gemini indisponible ({error}) : voix Edge utilisée à la place.")
if raw is None:
raw, spoken, used_voice = await edge.synthesize(text, voice, style, workdir)
processed = workdir / "voice.wav"
await audio.process_voice(raw, processed)
duration = (await proc.probe(processed)).duration
if not spoken:
# Gemini ne donne pas le minutage : Whisper le retrouve dans l'audio
spoken = await align.transcribe(processed, hint=text)
words = align.align_words(display_text, spoken, duration)
log.info("Voix prête : %s/%s, %.1f s, %d mots", used_engine, used_voice, duration, len(words))
return VoiceTrack(processed, duration, words, used_engine, used_voice, warnings)
+58
View File
@@ -0,0 +1,58 @@
"""Synthèse vocale Edge (gratuite, fournit le minutage des mots)."""
import logging
from pathlib import Path
import edge_tts
from .. import config
from ..align import Word
from ..voices import edge_fallbacks, resolve_edge_voice
log = logging.getLogger(__name__)
# Réglages de lecture par style (Edge n'interprète pas de consigne en texte)
STYLE_PROSODY = {
"dynamic": ("+8%", "+2Hz"),
"promo": ("+10%", "+3Hz"),
"calm": ("-8%", "-2Hz"),
"warm": ("-2%", "+0Hz"),
}
async def synthesize(
text: str, voice: str | None, style: str | None, workdir: Path
) -> tuple[Path, list[Word], str]:
"""Génère la voix ; renvoie le MP3, les mots minutés et la voix utilisée."""
rate, pitch = STYLE_PROSODY.get(style or "", ("+0%", "+0Hz"))
last_error: Exception | None = None
for candidate in edge_fallbacks(resolve_edge_voice(voice)):
target = workdir / "edge.mp3"
try:
communicate = edge_tts.Communicate(
text,
candidate.id,
rate=rate,
pitch=pitch,
# edge-tts 7 ne renvoie plus que les phrases par défaut
boundary="WordBoundary",
proxy=config.OUTBOUND_PROXY,
)
words: list[Word] = []
with target.open("wb") as out:
async for chunk in communicate.stream():
if chunk["type"] == "audio":
out.write(chunk["data"])
elif chunk["type"] == "WordBoundary":
start = chunk["offset"] / 10_000_000
words.append(Word(chunk["text"], start, start + chunk["duration"] / 10_000_000))
if target.stat().st_size == 0:
raise RuntimeError("audio vide")
log.info("Edge TTS : voix=%s, %d mots minutés", candidate.id, len(words))
return target, words, candidate.id
except Exception as error: # noqa: BLE001 — on essaie la voix suivante
log.warning("Edge TTS %s en échec : %s", candidate.id, error)
last_error = error
raise RuntimeError(f"Aucune voix Edge disponible : {last_error}")
+102
View File
@@ -0,0 +1,102 @@
"""Synthèse vocale Gemini (API generateContent, modalité audio)."""
import base64
import logging
import re
from pathlib import Path
import httpx
from .. import config, proc
from ..voices import resolve_gemini_voice, style_instruction
log = logging.getLogger(__name__)
API_ROOT = "https://generativelanguage.googleapis.com/v1beta/models"
FALLBACK_MODEL = "gemini-2.5-flash-preview-tts"
class GeminiError(RuntimeError):
pass
def build_prompt(text: str, style: str | None) -> str:
"""Texte envoyé au modèle : la consigne de style précède le texte à lire."""
instruction = style_instruction(style)
return f"{instruction} :\n{text}" if instruction else text
async def _request(model: str, prompt: str, voice: str, api_key: str) -> dict:
payload = {
"contents": [{"parts": [{"text": prompt}]}],
"generationConfig": {
"responseModalities": ["AUDIO"],
"speechConfig": {"voiceConfig": {"prebuiltVoiceConfig": {"voiceName": voice}}},
},
}
async with httpx.AsyncClient(timeout=120) as client:
# Clé en en-tête : dans l'URL, elle finissait dans les journaux
response = await client.post(
f"{API_ROOT}/{model}:generateContent",
json=payload,
headers={"x-goog-api-key": api_key},
)
if response.status_code != 200:
raise GeminiError(f"Gemini {model} : HTTP {response.status_code} {response.text[:300]}")
return response.json()
async def synthesize(
text: str, voice: str | None, style: str | None, api_key: str, workdir: Path
) -> tuple[Path, str]:
"""Génère la voix ; renvoie un WAV brut et le nom de la voix utilisée."""
gemini_voice = resolve_gemini_voice(voice).id
prompt = build_prompt(text, style)
model = config.GEMINI_TTS_MODEL
log.info("Gemini TTS : modèle=%s voix=%s style=%s (%d car.)", model, gemini_voice, style, len(text))
try:
data = await _request(model, prompt, gemini_voice, api_key)
except GeminiError as error:
if model == FALLBACK_MODEL or "HTTP 404" not in str(error):
raise
log.warning("Modèle %s indisponible, repli sur %s", model, FALLBACK_MODEL)
data = await _request(FALLBACK_MODEL, prompt, gemini_voice, api_key)
try:
part = next(p for p in data["candidates"][0]["content"]["parts"] if "inlineData" in p)["inlineData"]
except (KeyError, IndexError, StopIteration) as error:
raise GeminiError(f"Réponse Gemini sans audio : {str(data)[:300]}") from error
audio = base64.b64decode(part["data"])
mime = part.get("mimeType", "")
raw = workdir / "gemini_raw.wav"
if "wav" in mime:
raw.write_bytes(audio)
else:
# PCM brut 16 bits (« audio/L16;codec=pcm;rate=24000 »)
rate_match = re.search(r"rate=(\d+)", mime)
pcm = workdir / "gemini.pcm"
pcm.write_bytes(audio)
await proc.run(
[
"ffmpeg",
"-y",
"-hide_banner",
"-loglevel",
"error",
"-f",
"s16le",
"-ar",
rate_match.group(1) if rate_match else "24000",
"-ac",
"1",
"-i",
str(pcm),
str(raw),
],
timeout=60,
)
pcm.unlink(missing_ok=True)
return raw, gemini_voice
+131
View File
@@ -0,0 +1,131 @@
"""Catalogue des voix et des styles de lecture."""
from dataclasses import dataclass
@dataclass(frozen=True)
class Voice:
id: str
label: str
gender: str # "female" | "male"
# Voix natives de Gemini TTS (toutes multilingues, françaises comprises).
GEMINI_VOICES: tuple[Voice, ...] = (
Voice("Kore", "Kore — ferme", "female"),
Voice("Aoede", "Aoede — légère", "female"),
Voice("Leda", "Leda — jeune", "female"),
Voice("Zephyr", "Zephyr — lumineuse", "female"),
Voice("Callirrhoe", "Callirrhoe — décontractée", "female"),
Voice("Autonoe", "Autonoe — lumineuse", "female"),
Voice("Despina", "Despina — douce", "female"),
Voice("Erinome", "Erinome — claire", "female"),
Voice("Laomedeia", "Laomedeia — enjouée", "female"),
Voice("Achernar", "Achernar — tendre", "female"),
Voice("Gacrux", "Gacrux — mûre", "female"),
Voice("Pulcherrima", "Pulcherrima — affirmée", "female"),
Voice("Vindemiatrix", "Vindemiatrix — délicate", "female"),
Voice("Sulafat", "Sulafat — chaleureuse", "female"),
Voice("Charon", "Charon — informative", "male"),
Voice("Puck", "Puck — enjouée", "male"),
Voice("Fenrir", "Fenrir — enthousiaste", "male"),
Voice("Orus", "Orus — ferme", "male"),
Voice("Enceladus", "Enceladus — soufflée", "male"),
Voice("Iapetus", "Iapetus — claire", "male"),
Voice("Umbriel", "Umbriel — décontractée", "male"),
Voice("Algieba", "Algieba — veloutée", "male"),
Voice("Algenib", "Algenib — rocailleuse", "male"),
Voice("Rasalgethi", "Rasalgethi — informative", "male"),
Voice("Alnilam", "Alnilam — ferme", "male"),
Voice("Schedar", "Schedar — posée", "male"),
Voice("Achird", "Achird — amicale", "male"),
Voice("Zubenelgenubi", "Zubenelgenubi — naturelle", "male"),
Voice("Sadachbia", "Sadachbia — vive", "male"),
Voice("Sadaltager", "Sadaltager — experte", "male"),
)
EDGE_VOICES: tuple[Voice, ...] = (
Voice("fr-FR-VivienneMultilingualNeural", "Vivienne", "female"),
Voice("fr-FR-DeniseNeural", "Denise", "female"),
Voice("fr-FR-EloiseNeural", "Eloise", "female"),
Voice("fr-FR-RemyMultilingualNeural", "Rémy", "male"),
Voice("fr-FR-HenriNeural", "Henri", "male"),
)
# Consignes de lecture ajoutées au texte envoyé à Gemini.
STYLES: dict[str, tuple[str, str]] = {
"neutral": ("Neutre", ""),
"dynamic": (
"Dynamique",
"Lis ce texte en français sur un ton dynamique et enthousiaste, avec un rythme "
"entraînant, comme une vidéo courte sur les réseaux sociaux",
),
"warm": (
"Chaleureux",
"Lis ce texte en français sur un ton chaleureux, souriant et proche, comme si tu conseillais un ami",
),
"calm": (
"Calme",
"Lis ce texte en français sur un ton calme, posé et rassurant, sans te presser",
),
"promo": (
"Promo",
"Lis ce texte en français comme une annonce promotionnelle énergique, en "
"insistant sur les offres et les prix",
),
}
_GEMINI_BY_ID = {v.id.lower(): v for v in GEMINI_VOICES}
_EDGE_BY_ID = {v.id: v for v in EDGE_VOICES}
def resolve_gemini_voice(voice: str | None) -> Voice:
"""Voix Gemini à utiliser.
Accepte aussi les anciens identifiants « fr-FR-Standard-X » encore stockés
dans des jobs (B/D = voix masculine).
"""
if voice and voice.lower() in _GEMINI_BY_ID:
return _GEMINI_BY_ID[voice.lower()]
if voice and voice.endswith(("-B", "-D")) or voice == "male":
return _GEMINI_BY_ID["charon"]
return _GEMINI_BY_ID["kore"]
def voice_gender(voice: str | None) -> str:
"""Genre d'une voix, quel que soit le moteur d'origine."""
if not voice:
return "female"
if voice.lower() in _GEMINI_BY_ID:
return _GEMINI_BY_ID[voice.lower()].gender
if voice in _EDGE_BY_ID:
return _EDGE_BY_ID[voice].gender
if voice == "male" or voice.endswith(("-B", "-D")):
return "male"
return "male" if any(name in voice for name in ("Remy", "Henri", "Paul")) else "female"
def resolve_edge_voice(voice: str | None) -> Voice:
"""Voix Edge à utiliser ; une voix Gemini est remplacée par une voix Edge du même genre."""
if voice in _EDGE_BY_ID:
return _EDGE_BY_ID[voice]
gender = voice_gender(voice)
return next(v for v in EDGE_VOICES if v.gender == gender)
def edge_fallbacks(primary: Voice) -> list[Voice]:
"""Voix françaises de secours du même genre (jamais d'anglais)."""
same = [v for v in EDGE_VOICES if v.gender == primary.gender and v != primary]
return [primary, *same]
def style_instruction(style: str | None) -> str:
return STYLES.get(style or "neutral", STYLES["neutral"])[1]
def catalog() -> dict:
return {
"gemini": [v.__dict__ for v in GEMINI_VOICES],
"edge": [v.__dict__ for v in EDGE_VOICES],
"styles": [{"id": k, "label": v[0]} for k, v in STYLES.items()],
}
+5 -1385
View File
File diff suppressed because it is too large. Load diff
+3
View File
@@ -0,0 +1,3 @@
-r requirements.txt
pytest==8.3.5
ruff==0.11.8
+7 -9
View File
@@ -1,9 +1,7 @@
fastapi==0.109.0
uvicorn==0.27.0
python-multipart==0.0.6
requests==2.31.0
pydantic==2.6.0
edge-tts==6.1.12
emoji
ffsubsync==0.4.26
httpx>=0.25.0
fastapi==0.115.12
uvicorn[standard]==0.34.2
pydantic==2.11.4
httpx==0.28.1
edge-tts==7.2.8
emoji==2.14.1
faster-whisper==1.2.1
+9
View File
@@ -0,0 +1,9 @@
line-length = 110
target-version = "py311"
[lint]
select = ["E", "F", "W", "I", "B", "UP"]
ignore = [
"B008", # Depends() dans les signatures FastAPI
"B905", # zip() sur des séquences dont les longueurs sont liées par construction
]
-14
View File
@@ -1,14 +0,0 @@
[Script Info]
ScriptType: v4.00+
PlayResX: 1080
PlayResY: 1920
ScaledBorderAndShadow: yes
[V4+ Styles]
Format: Name, Fontname, Fontsize, PrimaryColour, SecondaryColour, OutlineColour, BackColour, Bold, Italic, Underline, StrikeOut, ScaleX, ScaleY, Spacing, Angle, BorderStyle, Outline, Shadow, Alignment, MarginL, MarginR, MarginV, Encoding
Style: Default,Sans,65,&H0000FFFF,&H00FFFFFF,&H00000000,&H80000000,-1,0,0,0,100,100,0,0,1,4,2,5,50,50,0,1
[Events]
Format: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text
Dialogue: 0,0:00:02.00,0:00:02.71,Default,,0,0,0,,{\kf31}Test {\kf31}test {\kf10}1
Dialogue: 0,0:00:02.71,0:00:02.87,Default,,0,0,0,,{\kf10}2 {\kf10}3
-13
View File
@@ -1,13 +0,0 @@
[Script Info]
ScriptType: v4.00+
PlayResX: 1080
PlayResY: 1920
ScaledBorderAndShadow: yes
[V4+ Styles]
Format: Name, Fontname, Fontsize, PrimaryColour, SecondaryColour, OutlineColour, BackColour, Bold, Italic, Underline, StrikeOut, ScaleX, ScaleY, Spacing, Angle, BorderStyle, Outline, Shadow, Alignment, MarginL, MarginR, MarginV, Encoding
Style: Default,Sans,65,&H00FFFFFF,&H000000FF,&H00000000,&H80000000,-1,0,0,0,100,100,0,0,1,4,2,5,50,50,0,1
[Events]
Format: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text
Dialogue: 0,0:00:02.00,0:00:04.00,Default,,0,0,0,,Hello World
View File
Whitespace-only changes.
+3
View File
@@ -0,0 +1,3 @@
import os
os.environ.setdefault("API_KEY", "test-key")
+31
View File
@@ -0,0 +1,31 @@
from app.align import Word, align_words, display_tokens, normalize
def test_normalize_strips_accents_and_punctuation():
assert normalize("Été,") == "ete"
assert normalize("«") == ""
def test_display_tokens_attach_isolated_punctuation():
assert display_tokens("Bonjour ! Venez vite .") == ["Bonjour!", "Venez", "vite."]
def test_align_uses_engine_timings_and_keeps_display_text():
spoken = [Word("Bonjour", 0.05, 0.6), Word("découvrez", 0.99, 1.35), Word("nos", 1.35, 1.44)]
words = align_words("Bonjour, découvrez nos", spoken)
assert [w.text for w in words] == ["Bonjour,", "découvrez", "nos"]
assert words[1].start == 0.99
def test_unmatched_words_are_interpolated_between_neighbours():
spoken = [Word("les", 0.0, 0.2), Word("prix", 1.0, 1.3)]
words = align_words("les super promos prix", spoken)
assert [w.text for w in words] == ["les", "super", "promos", "prix"]
assert 0.2 <= words[1].start < words[2].start < 1.0
assert all(b.start > a.start for a, b in zip(words, words[1:]))
def test_without_spoken_words_text_is_spread_over_duration():
words = align_words("un deux trois", [], total_duration=3.0)
assert words[0].start == 0.0
assert abs(words[-1].end - 3.0) < 1e-6
+77
View File
@@ -0,0 +1,77 @@
from pathlib import Path
from app.render import RenderPlan, build_command
def _graph(cmd):
return cmd[cmd.index("-filter_complex") + 1]
def test_voice_longer_than_video_extends_with_frozen_frame():
plan = RenderPlan(
video=Path("in.mp4"),
video_duration=5.0,
output=Path("out.mp4"),
voice=Path("v.wav"),
voice_duration=10.0,
)
assert plan.total_duration == 12.8
cmd = build_command(plan)
assert "tpad=stop_mode=clone:stop_duration=7.800" in _graph(cmd)
assert cmd[cmd.index("-t") + 1] == "12.800"
def test_music_is_looped_and_ducked_under_voice():
plan = RenderPlan(
video=Path("in.mp4"),
video_duration=20.0,
output=Path("out.mp4"),
music=Path("m.mp3"),
voice=Path("v.wav"),
voice_duration=5.0,
)
cmd = build_command(plan)
assert cmd[cmd.index("m.mp3") - 3 : cmd.index("m.mp3")] == ["-stream_loop", "-1", "-i"]
graph = _graph(cmd)
assert "sidechaincompress" in graph
assert "amix=inputs=2:duration=longest" in graph
assert "loudnorm=I=-14" in graph
def test_no_brightness_hack_and_hdr_tonemapped():
plan = RenderPlan(video=Path("in.mp4"), video_duration=5.0, output=Path("out.mp4"), is_hdr=True)
graph = _graph(build_command(plan))
assert "eq=" not in graph
assert "tonemap=tonemap=hable" in graph
assert "flags=lanczos" in graph
def test_original_audio_kept_when_nothing_added():
plan = RenderPlan(
video=Path("in.mp4"), video_duration=5.0, output=Path("out.mp4"), keep_original_audio=True
)
cmd = build_command(plan)
assert cmd[cmd.index("-map", cmd.index("[vout]")) + 1] == "0:a:0"
def test_encoding_targets_social_networks():
cmd = build_command(RenderPlan(video=Path("in.mp4"), video_duration=5.0, output=Path("out.mp4")))
joined = " ".join(cmd)
assert "-crf 19" in joined and "-b:v" not in joined
assert "-ar 48000" in joined and "-b:a 192k" in joined
assert "-colorspace bt709" in joined
def test_big_logo_waits_for_the_end_of_the_voice():
plan = RenderPlan(
video=Path("in.mp4"),
video_duration=6.0,
output=Path("out.mp4"),
voice=Path("v.wav"),
voice_duration=4.8,
watermark=Path("logo.png"),
outro=Path("o.ass"),
)
assert plan.logo_start == 6.8
assert plan.total_duration == 9.3
assert "gte(t,6.800)" in _graph(build_command(plan))
+23
View File
@@ -0,0 +1,23 @@
from app.align import Word
from app.subtitles import ass_time, group_lines, write_captions
def test_ass_time_format():
assert ass_time(0) == "0:00:00.00"
assert ass_time(61.234) == "0:01:01.23"
def test_lines_break_on_punctuation_and_length():
words = [Word(t, i, i + 0.5) for i, t in enumerate("Bonjour, venez découvrir nos nouveautés".split())]
lines = group_lines(words)
assert [w.text for w in lines[0]] == ["Bonjour,"]
assert all(len(line) <= 3 for line in lines)
def test_captions_are_offset_and_escape_braces(tmp_path):
path = tmp_path / "c.ass"
write_captions([Word("{prix}", 0.0, 0.5), Word("fous", 0.5, 1.0)], path, offset=2.0, font_size=80)
content = path.read_text()
assert "Dialogue: 0,0:00:02.00," in content
assert "{prix}" not in content.split("[Events]")[1].replace("{\\", "")
assert "(prix)" in content
+23
View File
@@ -0,0 +1,23 @@
from app.tts.gemini import build_prompt
from app.voices import GEMINI_VOICES, edge_fallbacks, resolve_edge_voice, resolve_gemini_voice
def test_thirty_gemini_voices():
assert len(GEMINI_VOICES) == 30
def test_legacy_gemini_identifiers_still_resolve():
assert resolve_gemini_voice("fr-FR-Standard-B").id == "Charon"
assert resolve_gemini_voice("fr-FR-Standard-A").id == "Kore"
assert resolve_gemini_voice("puck").id == "Puck"
def test_edge_fallbacks_are_french_and_same_gender():
voice = resolve_edge_voice("Fenrir") # voix Gemini masculine
assert voice.gender == "male"
assert all(v.id.startswith("fr-FR") and v.gender == "male" for v in edge_fallbacks(voice))
def test_style_prompt_precedes_text():
assert build_prompt("Salut", "neutral") == "Salut"
assert build_prompt("Salut", "dynamic").endswith(":\nSalut")
+32 -102
View File
@@ -5,14 +5,13 @@
import { Router, Request, Response } from 'express';
import type { User } from '@shared/schema';
import { storage } from '../storage';
import { ffmpegService } from '../services/ffmpeg';
import { resolveInternalUrl } from '../services/minio';
import { videoReelParamsSchema, type VideoReelParams } from '@shared/reel';
import { ffmpegService, FFmpegServiceError } from '../services/ffmpeg';
import { ttsPreviewSchema, videoReelParamsSchema, type VideoReelParams } from '@shared/reel';
import { enqueueReelJob, countActiveReelJobs } from '../services/reels/queue';
import { resolveGeminiApiKey, resolveLogoPath, resolveMusicUrl, resolveStoreName } from '../services/reels/assets';
import { resolveGeminiApiKey, resolveStoreName } from '../services/reels/assets';
import { openRouterService, describeGenerationError } from '../services/openrouter';
import { ttsSyncService } from '../services/ttsSync';
import { estimateVoiceTiming } from '../services/ttsSync';
/** Piste musicale telle qu'attendue par le client. */
interface MusicTrack {
id: string;
@@ -270,105 +269,37 @@ reelsRouter.post('/reels/generate-text', async (req: Request, res: Response) =>
}
});
/**
* Prévisualiser un Reel (traitement sans publication)
* POST /api/reels/preview
*/
reelsRouter.post('/reels/preview', async (req: Request, res: Response) => {
try {
const user = req.user as User;
const {
videoMediaId,
musicTrackId,
musicUrl,
overlayText,
ttsEnabled,
ttsVoice,
ttsEngine,
wordDuration = 0.6,
fontSize = 64,
musicVolume = 0.25,
drawText = true,
stabilize = false,
enableEndingEffect = true,
} = req.body;
// Récupérer le média vidéo
const media = await storage.getMediaById(videoMediaId);
if (!media) {
return res.status(404).json({ error: 'Vidéo non trouvée' });
}
if (media.type !== 'video') {
return res.status(400).json({ error: 'Le média doit être une vidéo' });
}
const [finalMusicUrl, logoPath, geminiApiKey] = await Promise.all([
resolveMusicUrl(musicTrackId, musicUrl),
resolveLogoPath(),
resolveGeminiApiKey(ttsEngine),
]);
const watermarkUrl = logoPath ? resolveInternalUrl(logoPath) : undefined;
const finalWordDuration = wordDuration;
// Traiter la vidéo via FFmpeg
const result = await ffmpegService.processReelFromUrl(resolveInternalUrl(media.originalUrl), {
text: overlayText,
musicUrl: finalMusicUrl,
ttsEnabled,
ttsVoice,
ttsEngine,
geminiApiKey,
wordDuration: finalWordDuration,
fontSize,
musicVolume,
drawText,
stabilize,
watermarkUrl,
enableEndingEffect,
});
if (!result.success) {
return res.status(500).json({ error: result.error || 'Erreur de traitement vidéo' });
}
// Retourner la vidéo en base64 pour prévisualisation
res.json({
success: true,
videoBase64: result.videoBase64,
duration: result.duration,
});
} catch (error) {
console.error('❌ Error previewing Reel:', error);
res.status(500).json({ error: 'Erreur lors de la prévisualisation du Reel' });
}
});
/**
* Prévisualiser la voix TTS
* POST /api/reels/tts-preview
*/
reelsRouter.post('/reels/tts-preview', async (req: Request, res: Response) => {
const parsed = ttsPreviewSchema.safeParse(req.body);
if (!parsed.success) {
return res.status(400).json({ error: parsed.error.issues[0]?.message ?? 'Paramètres invalides' });
}
const { text, ttsVoice, ttsEngine, ttsStyle } = parsed.data;
try {
const user = req.user as User;
const { text, ttsVoice, ttsEngine } = req.body;
if (!text) {
return res.status(400).json({ error: 'Texte requis' });
}
const geminiApiKey = await resolveGeminiApiKey(ttsEngine);
const result = await ffmpegService.previewTTS(text, ttsVoice, ttsEngine, geminiApiKey);
if (!result.success) {
return res.status(500).json({ error: result.error || 'Erreur de génération TTS' });
}
res.json({ success: true, audioBase64: result.audioBase64 });
const preview = await ffmpegService.previewVoice(text, {
voice: ttsVoice,
engine: ttsEngine,
style: ttsStyle,
geminiApiKey: await resolveGeminiApiKey(ttsEngine),
});
res.json({
success: true,
audioBase64: preview.audio.toString('base64'),
duration: preview.duration,
words: preview.words,
engine: preview.engine,
voice: preview.voice,
warnings: preview.warnings,
});
} catch (error) {
console.error('❌ Error generating TTS preview:', error);
res.status(500).json({ error: 'Erreur lors de la génération de la voix' });
const message = error instanceof Error ? error.message : 'Erreur lors de la génération de la voix';
res.status(error instanceof FFmpegServiceError && error.status === 400 ? 400 : 502).json({ error: message });
}
});
@@ -378,13 +309,12 @@ reelsRouter.post('/reels/tts-preview', async (req: Request, res: Response) => {
*/
reelsRouter.post('/reels/sync-info', async (req: Request, res: Response) => {
try {
const { text, ttsVoice, ttsEngine } = req.body;
if (!text || !ttsVoice) {
return res.status(400).json({ error: 'Texte et voix requis' });
const { text } = req.body;
if (!text) {
return res.status(400).json({ error: 'Texte requis' });
}
const geminiApiKey = await resolveGeminiApiKey(ttsEngine);
const sync = await ttsSyncService.calculateSyncTiming(text, ttsVoice, ttsEngine, geminiApiKey);
res.json(sync);
// Estimation locale : ne déclenche aucune synthèse (payante avec Gemini)
res.json(estimateVoiceTiming(text));
} catch (error) {
console.error('❌ Error calculating sync:', error);
res.status(500).json({ error: 'Erreur de calcul de synchronisation' });
+2
View File
@@ -70,8 +70,10 @@ remotionRouter.post("/render", upload.fields([{ name: "images", maxCount: 4 }, {
overlayText: req.body.overlayText,
musicUrl: musicFile ? `/uploads/temp/${path.basename(musicFile.path)}` : req.body.musicTrackUrl,
musicVolume: Number.isFinite(musicVolume) ? musicVolume : undefined,
ttsEnabled: req.body.ttsEnabled !== "false",
ttsEngine: req.body.ttsEngine || undefined,
ttsVoice: req.body.ttsVoice,
ttsStyle: req.body.ttsStyle || undefined,
storeName: await resolveStoreName(user.id, req.body.selectedPageId),
tempFiles: uploaded.map((f) => f.path),
});
+144 -289
View File
@@ -1,214 +1,120 @@
/**
* FFmpeg Docker API Service
*
* Intégration avec l'API FFmpeg Docker locale pour le traitement vidéo des Reels.
* L'API attend une vidéo en base64 et retourne la vidéo traitée en base64.
* Client du service FFmpeg (conteneur Python `ffmpeg-service`).
*
* Le service rend le Reel puis expose le MP4 en téléchargement
* (GET /files/{job}/output.mp4) : la vidéo ne transite plus en base64 dans du
* JSON, qui gonflait sa taille d'un tiers et la gardait entière en mémoire.
*/
interface FFmpegReelRequest {
video_base64?: string; // Vidéo source en base64
video_url?: string; // OU URL de la vidéo source
text?: string; // Texte overlay style TikTok
music_id?: string; // ID de la musique (catalogue FFmpeg)
music_url?: string; // OU URL directe de la musique
tts_enabled?: boolean; // Activation du TTS
tts_voice?: string; // Voix TTS (ex: fr-FR-VivienneNeural)
tts_engine?: string; // Moteur TTS: "edge" ou "gemini"
gemini_api_key?: string; // Clé API Google Gemini pour TTS
word_duration?: number; // Durée par mot (default: 0.6s)
font_size?: number; // Taille police (default: 24)
music_volume?: number; // Volume musique (default: 0.25)
draw_text?: boolean; // Dessiner le texte sur la vidéo (default: true)
stabilize?: boolean; // Stabilisation vidéo via vidstab (default: false)
watermark_url?: string; // URL du logo
store_name?: string; // Nom du magasin pour l'outro
enable_ending_effect?: boolean; // Activer l'effet de fin (logo+fondu)
}
interface FFmpegReelResponse {
success: boolean;
output_base64?: string;
duration?: number;
detail?: string;
tts_error?: string;
}
import type { TtsEngine, TtsStyle } from '@shared/voices';
/** Un rendu long (stabilisation + encodage) peut dépasser plusieurs minutes. */
const PROCESS_TIMEOUT_MS = 15 * 60_000;
const TTS_TIMEOUT_MS = 2 * 60_000;
const PROCESS_TIMEOUT_MS = 20 * 60_000;
const DOWNLOAD_TIMEOUT_MS = 5 * 60_000;
const TTS_TIMEOUT_MS = 3 * 60_000;
const HEALTH_TIMEOUT_MS = 5_000;
export interface ReelRenderOptions {
text?: string;
musicUrl?: string;
ttsEnabled?: boolean;
ttsVoice?: string;
ttsEngine?: TtsEngine;
ttsStyle?: TtsStyle;
geminiApiKey?: string;
fontSize?: number;
musicVolume?: number;
drawText?: boolean;
stabilize?: boolean;
watermarkUrl?: string;
storeName?: string;
enableEndingEffect?: boolean;
}
export interface ReelRenderResult {
video: Buffer;
duration: number;
ttsEngine?: string | null;
ttsVoice?: string | null;
warnings: string[];
}
export interface TimedWord {
text: string;
start: number;
end: number;
}
export interface VoicePreview {
audio: Buffer;
duration: number;
words: TimedWord[];
engine: string;
voice: string;
warnings: string[];
}
interface FFmpegConfig {
apiUrl: string;
apiKey: string;
}
/** Erreur renvoyée par le service, avec son message lisible. */
export class FFmpegServiceError extends Error {
constructor(message: string, readonly status?: number) {
super(message);
this.name = 'FFmpegServiceError';
}
}
export class FFmpegService {
private config: FFmpegConfig | null = null;
/**
* Configure le service avec l'URL et la clé API
*/
configure(apiUrl: string, apiKey: string): void {
this.config = { apiUrl, apiKey };
console.log('🎬 FFmpeg Service configured:', apiUrl);
this.config = { apiUrl: apiUrl.replace(/\/$/, ''), apiKey };
console.log('🎬 FFmpeg Service configured:', this.config.apiUrl);
}
/**
* Vérifie que le service est configuré
*/
private ensureConfigured(): FFmpegConfig {
if (!this.config) {
throw new Error('FFmpeg Service not configured. Call configure() first.');
throw new FFmpegServiceError("Le service FFmpeg n'est pas configuré (FFMPEG_API_URL / FFMPEG_API_KEY).");
}
return this.config;
}
/**
* Traite une vidéo pour créer un Reel avec musique et texte overlay
*
* @param videoBase64 - Vidéo source encodée en base64
* @param options - Options de traitement (texte, musique, etc.)
* @returns Vidéo traitée en base64
*/
async processReelVideo(
videoBase64: string,
options: {
text?: string;
musicId?: string;
musicUrl?: string;
ttsEnabled?: boolean;
ttsVoice?: string;
ttsEngine?: string;
geminiApiKey?: string;
wordDuration?: number;
fontSize?: number;
musicVolume?: number;
drawText?: boolean;
stabilize?: boolean;
watermarkUrl?: string;
storeName?: string;
enableEndingEffect?: boolean;
} = {}
): Promise<{ success: boolean; videoBase64?: string; duration?: number; error?: string }> {
private async call(path: string, init: RequestInit & { timeoutMs: number }): Promise<Response> {
const config = this.ensureConfigured();
const requestBody: FFmpegReelRequest = {
video_base64: videoBase64,
text: options.text,
music_id: options.musicId,
music_url: options.musicUrl,
tts_enabled: options.ttsEnabled,
tts_voice: options.ttsVoice,
tts_engine: options.ttsEngine,
gemini_api_key: options.geminiApiKey,
word_duration: options.wordDuration ?? 0.6,
font_size: options.fontSize ?? 64,
music_volume: options.musicVolume ?? 0.25,
draw_text: options.drawText ?? true,
stabilize: options.stabilize ?? false,
watermark_url: options.watermarkUrl,
store_name: options.storeName,
enable_ending_effect: options.enableEndingEffect ?? true,
};
// Remove undefined values
Object.keys(requestBody).forEach(key => {
if (requestBody[key as keyof FFmpegReelRequest] === undefined) {
delete requestBody[key as keyof FFmpegReelRequest];
}
const { timeoutMs, headers, ...rest } = init;
const response = await fetch(`${config.apiUrl}${path}`, {
...rest,
headers: { 'X-API-Key': config.apiKey, ...headers },
signal: AbortSignal.timeout(timeoutMs),
});
console.log('🎬 Processing Reel video:', {
hasVideo: !!videoBase64,
hasText: !!options.text,
hasMusicId: !!options.musicId,
hasMusicUrl: !!options.musicUrl,
hasTTS: options.ttsEnabled,
drawText: options.drawText,
});
try {
const response = await fetch(`${config.apiUrl}/process-reel`, {
method: 'POST',
headers: {
'Content-Type': 'application/json',
'X-API-Key': config.apiKey,
},
body: JSON.stringify(requestBody),
signal: AbortSignal.timeout(PROCESS_TIMEOUT_MS),
});
if (!response.ok) {
const errorText = await response.text();
console.error('❌ FFmpeg API error:', response.status, errorText);
return {
success: false,
error: `FFmpeg API error: ${response.status} - ${errorText}`,
};
}
const data = await response.json() as FFmpegReelResponse;
if (!data.success) {
console.error('❌ FFmpeg processing failed:', data.detail);
return {
success: false,
error: data.detail || 'Unknown FFmpeg processing error',
};
}
console.log('✅ Reel video processed successfully, duration:', data.duration);
return {
success: true,
videoBase64: data.output_base64,
duration: data.duration,
};
} catch (error) {
console.error('❌ FFmpeg Service error:', error);
return {
success: false,
error: error instanceof Error ? error.message : 'Unknown error',
};
if (!response.ok) {
const body = await response.text();
let detail = body;
try {
detail = JSON.parse(body).detail ?? body;
} catch { /* corps non JSON */ }
throw new FFmpegServiceError(String(detail).slice(0, 2000), response.status);
}
return response;
}
/**
* Traite une vidéo depuis une URL (télécharge, traite, retourne base64)
* Rend un Reel à partir de l'URL d'une vidéo et renvoie le MP4 produit.
* Lève FFmpegServiceError en cas d'échec (voix comprise).
*/
async processReelFromUrl(
videoUrl: string,
options: {
text?: string;
musicId?: string;
musicUrl?: string;
ttsEnabled?: boolean;
ttsVoice?: string;
ttsEngine?: string;
geminiApiKey?: string;
wordDuration?: number;
fontSize?: number;
musicVolume?: number;
drawText?: boolean;
stabilize?: boolean;
watermarkUrl?: string;
storeName?: string;
enableEndingEffect?: boolean;
} = {}
): Promise<{ success: boolean; videoBase64?: string; duration?: number; error?: string; ttsError?: string }> {
const config = this.ensureConfigured();
const requestBody: FFmpegReelRequest = {
async renderReel(videoUrl: string, options: ReelRenderOptions = {}): Promise<ReelRenderResult> {
const body = {
video_url: videoUrl,
text: options.text,
music_id: options.musicId,
music_url: options.musicUrl,
tts_enabled: options.ttsEnabled,
tts_enabled: options.ttsEnabled ?? false,
tts_voice: options.ttsVoice,
tts_engine: options.ttsEngine,
tts_style: options.ttsStyle,
gemini_api_key: options.geminiApiKey,
word_duration: options.wordDuration ?? 0.6,
font_size: options.fontSize ?? 64,
music_volume: options.musicVolume ?? 0.25,
draw_text: options.drawText ?? true,
@@ -218,139 +124,88 @@ export class FFmpegService {
enable_ending_effect: options.enableEndingEffect ?? true,
};
// Remove undefined values
Object.keys(requestBody).forEach(key => {
if (requestBody[key as keyof FFmpegReelRequest] === undefined) {
delete requestBody[key as keyof FFmpegReelRequest];
}
});
console.log('🎬 Processing Reel from URL:', {
console.log('🎬 Rendu du Reel :', {
videoUrl,
hasText: !!options.text,
textLength: options.text?.length || 0,
hasMusicId: !!options.musicId,
hasMusicUrl: !!options.musicUrl,
ttsEnabled: options.ttsEnabled,
drawText: options.drawText,
textLength: options.text?.length ?? 0,
music: !!options.musicUrl,
tts: options.ttsEnabled ? `${options.ttsEngine}/${options.ttsVoice}/${options.ttsStyle ?? 'neutral'}` : false,
});
const debugBody = { ...requestBody };
console.log('📤 Sending to FFmpeg API:', JSON.stringify({ ...debugBody, text: debugBody.text ? `[${debugBody.text.length} chars]` : undefined }));
const response = await this.call('/process-reel', {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify(body),
timeoutMs: PROCESS_TIMEOUT_MS,
});
const data = await response.json() as {
job_id: string;
output_path: string;
duration: number;
tts_engine?: string | null;
tts_voice?: string | null;
warnings?: string[];
};
try {
const response = await fetch(`${config.apiUrl}/process-reel`, {
method: 'POST',
headers: {
'Content-Type': 'application/json',
'X-API-Key': config.apiKey,
},
body: JSON.stringify(requestBody),
signal: AbortSignal.timeout(PROCESS_TIMEOUT_MS),
});
if (!response.ok) {
const errorText = await response.text();
console.error('❌ FFmpeg API error:', response.status, errorText);
return {
success: false,
error: `FFmpeg API error: ${response.status} - ${errorText}`,
};
}
const data = await response.json() as FFmpegReelResponse;
if (!data.success) {
console.error('❌ FFmpeg processing failed:', data.detail);
return {
success: false,
error: data.detail || 'Unknown FFmpeg processing error',
};
}
if (data.tts_error) {
console.error('❌ TTS failed in Python service:', data.tts_error);
}
console.log('✅ Reel video processed successfully from URL');
const file = await this.call(data.output_path, { method: 'GET', timeoutMs: DOWNLOAD_TIMEOUT_MS });
const video = Buffer.from(await file.arrayBuffer());
for (const warning of data.warnings ?? []) console.warn(`⚠️ [FFmpeg] ${warning}`);
return {
success: true,
videoBase64: data.output_base64,
video,
duration: data.duration,
ttsError: data.tts_error,
};
} catch (error) {
console.error('❌ FFmpeg Service error:', error);
return {
success: false,
error: error instanceof Error ? error.message : 'Unknown error',
ttsEngine: data.tts_engine,
ttsVoice: data.tts_voice,
warnings: data.warnings ?? [],
};
} finally {
// Le fichier n'est plus utile au service une fois récupéré
this.call(`/jobs/${data.job_id}`, { method: 'DELETE', timeoutMs: HEALTH_TIMEOUT_MS })
.catch(() => { /* purgé plus tard par le service */ });
}
}
/**
* Vérifie la santé de l'API FFmpeg
*/
async healthCheck(): Promise<boolean> {
try {
const config = this.ensureConfigured();
const response = await fetch(`${config.apiUrl}/health`, {
method: 'GET',
headers: {
'X-API-Key': config.apiKey,
},
signal: AbortSignal.timeout(HEALTH_TIMEOUT_MS),
});
return response.ok;
await this.call('/health', { method: 'GET', timeoutMs: HEALTH_TIMEOUT_MS });
return true;
} catch {
return false;
}
}
async previewTTS(
/** Génère la voix seule (aperçu), avec le minutage de chaque mot. */
async previewVoice(
text: string,
ttsVoice?: string,
ttsEngine?: string,
geminiApiKey?: string
): Promise<{ success: boolean; audioBase64?: string; error?: string }> {
const config = this.ensureConfigured();
try {
const response = await fetch(`${config.apiUrl}/preview-tts`, {
method: 'POST',
headers: {
'Content-Type': 'application/json',
'X-API-Key': config.apiKey,
},
body: JSON.stringify({
text,
tts_enabled: true,
tts_voice: ttsVoice,
tts_engine: ttsEngine,
gemini_api_key: geminiApiKey,
}),
signal: AbortSignal.timeout(TTS_TIMEOUT_MS),
});
if (!response.ok) {
const errorText = await response.text();
return { success: false, error: `FFmpeg API error: ${response.status} - ${errorText}` };
}
const data = await response.json();
if (!data.success) {
return { success: false, error: data.detail };
}
return { success: true, audioBase64: data.audio_base64 };
} catch (error) {
console.error('❌ TTS Preview error:', error);
return {
success: false,
error: error instanceof Error ? error.message : 'Unknown error',
};
}
options: { voice?: string; engine?: TtsEngine; style?: TtsStyle; geminiApiKey?: string } = {},
): Promise<VoicePreview> {
const response = await this.call('/preview-tts', {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({
text,
tts_voice: options.voice,
tts_engine: options.engine,
tts_style: options.style,
gemini_api_key: options.geminiApiKey,
}),
timeoutMs: TTS_TIMEOUT_MS,
});
const data = await response.json() as {
audio_base64: string;
duration: number;
words: TimedWord[];
engine: string;
voice: string;
warnings?: string[];
};
return {
audio: Buffer.from(data.audio_base64, 'base64'),
duration: data.duration,
words: data.words ?? [],
engine: data.engine,
voice: data.voice,
warnings: data.warnings ?? [],
};
}
}
+34 -52
View File
@@ -6,11 +6,10 @@
import fs from "fs";
import path from "path";
import * as musicMetadata from "music-metadata";
import { bundle } from "@remotion/bundler";
import { renderMedia, selectComposition } from "@remotion/renderer";
import { imagesReelParamsSchema, type ImagesReelResult } from "@shared/reel";
import { ffmpegService } from "../ffmpeg";
import { ffmpegService, type TimedWord } from "../ffmpeg";
import { generateVideoThumbnail } from "../thumbnail";
import type { JobContext } from "./queue";
import { resolveGeminiApiKey, resolveLogoPath } from "./assets";
@@ -88,43 +87,31 @@ export function stripForTTS(text: string): string {
.trim();
}
/** Nombre de syllabes d'un mot français (groupes de voyelles). */
function countSyllablesFr(word: string): number {
const clean = word.replace(/[^a-zàâéèêëîïôùûüç]/gi, "").toLowerCase();
if (!clean) return 1;
return Math.max(1, clean.match(/[aeiouyàâéèêëîïôùûü]+/gi)?.length ?? 1);
const EDGE_PUNCT = /^[.,!?;:…«»"'()\[\]]+|[.,!?;:…«»"'()\[\]]+$/g;
export interface WordTiming {
word: string;
startFrame: number;
endFrame: number;
}
const PUNCT_ONLY = /^[.,!?;:…\-—«»"''()\[\]]+$/;
const EDGE_PUNCT = /^[.,!?;:…«»"''()\[\]]+|[.,!?;:…«»"''()\[\]]+$/g;
/**
* Timings des mots prononcés (en images), répartis au prorata des syllabes.
* Estimation provisoire : remplacée par les vrais timings de la voix au lot
* « voix et sous-titres ».
* Convertit le minutage réel des mots (en secondes, fourni par la voix) en
* images. Chaque mot reste affiché jusqu'au début du suivant : pas de trou
* pendant les respirations.
*/
export function computeWordTimings(
displayText: string,
audioDurationSeconds: number,
fps: number,
startFrame: number,
): Array<{ word: string; startFrame: number; endFrame: number }> {
const spokenWords = stripForTTS(displayText)
.split(/\s+/)
.filter((w) => w && !PUNCT_ONLY.test(w));
if (spokenWords.length === 0) return [];
const cleanWords = spokenWords.map((w) => w.replace(EDGE_PUNCT, "") || w);
const syllables = cleanWords.map(countSyllablesFr);
const totalSyllables = syllables.reduce((a, b) => a + b, 0);
const totalFrames = audioDurationSeconds * fps;
let currentFrame = startFrame;
return cleanWords.map((word, i) => {
const wordStart = currentFrame;
currentFrame += Math.round((syllables[i] / totalSyllables) * totalFrames);
return { word, startFrame: wordStart, endFrame: currentFrame };
});
export function toWordTimings(words: TimedWord[], fps: number): WordTiming[] {
const timings = words
.map((w) => ({ word: w.text.replace(EDGE_PUNCT, "") || w.text, start: w.start, end: w.end }))
.filter((w) => w.word.trim());
return timings.map((w, i) => ({
word: w.word,
startFrame: Math.round(w.start * fps),
endFrame: Math.max(
Math.round(w.start * fps) + 1,
Math.round((i + 1 < timings.length ? timings[i + 1].start : w.end) * fps),
),
}));
}
export async function runImagesReelJob({ job, progress }: JobContext): Promise<ImagesReelResult> {
@@ -141,32 +128,27 @@ export async function runImagesReelJob({ job, progress }: JobContext): Promise<I
// --- Voix ---
let audioUrl: string | undefined;
let wordTimings: ReturnType<typeof computeWordTimings> | undefined;
let wordTimings: WordTiming[] | undefined;
let audioDuration = 0;
const ttsText = overlayText ? stripForTTS(overlayText) : "";
if (overlayText && ttsText) {
if (params.ttsEnabled && overlayText && ttsText) {
await progress(15, "voice");
const geminiApiKey = await resolveGeminiApiKey(params.ttsEngine);
const tts = await ffmpegService.previewTTS(ttsText, params.ttsVoice, params.ttsEngine, geminiApiKey);
if (!tts.success || !tts.audioBase64) {
throw new Error(`La voix n'a pas pu être générée : ${tts.error ?? "réponse vide"}`);
}
const voice = await ffmpegService.previewVoice(ttsText, {
voice: params.ttsVoice,
engine: params.ttsEngine,
style: params.ttsStyle,
geminiApiKey: await resolveGeminiApiKey(params.ttsEngine),
});
for (const warning of voice.warnings) console.warn(`⚠️ [Reels] ${warning}`);
const audioFilename = `tts-${job.id}.mp3`;
const audioPath = path.join(REMOTION_TEMP_DIR, audioFilename);
const audioBuffer = Buffer.from(tts.audioBase64, "base64");
await fs.promises.writeFile(audioPath, audioBuffer);
await fs.promises.writeFile(audioPath, voice.audio);
jobTempFiles.push(audioPath);
audioUrl = localHttpUrl(`/uploads/temp/${audioFilename}`);
try {
audioDuration = (await musicMetadata.parseFile(audioPath)).format.duration ?? 0;
} catch {
audioDuration = audioBuffer.length / 16000; // estimation à 128 kb/s
}
audioDuration = Math.max(audioDuration, ttsText.split(/\s+/).length * 0.35);
wordTimings = computeWordTimings(overlayText, audioDuration, FPS, 0);
audioDuration = voice.duration;
wordTimings = toWordTimings(voice.words, FPS);
}
// --- Durée : 25 à 30 s ---
+1
View File
@@ -94,6 +94,7 @@ export async function startReelWorker(): Promise<void> {
console.error("❌ [ReelQueue] Récupération des jobs orphelins impossible :", error);
}
console.log(`🎬 [ReelQueue] Worker démarré (traitements : ${Array.from(handlers.keys()).join(", ")})`);
pollTimer = setInterval(kick, POLL_INTERVAL_MS);
pollTimer.unref();
kick();
+4 -12
View File
@@ -30,14 +30,14 @@ export async function runVideoReelJob({ job, progress }: JobContext) {
await progress(15, "render");
const startedAt = Date.now();
const rendered = await ffmpegService.processReelFromUrl(resolveInternalUrl(media.originalUrl), {
const rendered = await ffmpegService.renderReel(resolveInternalUrl(media.originalUrl), {
text: params.overlayText,
musicUrl,
ttsEnabled: params.ttsEnabled,
ttsVoice: params.ttsVoice,
ttsEngine: params.ttsEngine,
ttsStyle: params.ttsStyle,
geminiApiKey,
wordDuration: params.wordDuration,
fontSize: params.fontSize,
musicVolume: params.musicVolume,
drawText: params.drawText,
@@ -46,18 +46,10 @@ export async function runVideoReelJob({ job, progress }: JobContext) {
storeName: params.storeName,
enableEndingEffect: params.enableEndingEffect,
});
console.log(`⏱️ [Reels] FFmpeg : ${((Date.now() - startedAt) / 1000).toFixed(1)} s`);
if (!rendered.success || !rendered.videoBase64) {
throw new Error(rendered.error || "Erreur de traitement vidéo FFmpeg");
}
if (rendered.ttsError) {
// Voix demandée mais absente : ne jamais publier un Reel muet sans le dire
throw new Error(`La voix n'a pas pu être générée : ${rendered.ttsError}`);
}
console.log(`⏱️ [Reels] Rendu : ${((Date.now() - startedAt) / 1000).toFixed(1)} s, vidéo de ${rendered.duration.toFixed(1)} s`);
await progress(65, "store");
const videoBuffer = Buffer.from(rendered.videoBase64, "base64");
const videoBuffer = rendered.video;
const processedMedia = await storeRenderedVideo(job.userId, videoBuffer, `reel-${Date.now()}.mp4`);
await storage.updatePostMedia(postId, [processedMedia.id]);
+35 -62
View File
@@ -1,5 +1,15 @@
import { ffmpegService } from './ffmpeg';
import * as musicMetadata from 'music-metadata';
/**
* Estimation de la durée de lecture d'un texte, affichée pendant la saisie.
*
* Auparavant, chaque pause de frappe générait une voix complète pour la
* mesurer (un appel Gemini facturé à chaque fois). Le minutage réel des mots
* vient désormais de la voix au moment du rendu ; ici, une estimation suffit.
*/
/** Débit moyen d'une voix de synthèse française, en mots par seconde. */
const WORDS_PER_SECOND = 2.6;
/** Au-delà, la voix dépasse la durée confortable d'un Reel. */
const MAX_COMFORTABLE_SECONDS = 45;
export interface SyncTiming {
wordDuration: number;
@@ -10,67 +20,30 @@ export interface SyncTiming {
warnings: string[];
}
export class TtsSyncService {
/**
* Calcule le word_duration optimal pour synchroniser l'affichage du texte
* avec la durée réelle de la voix TTS générée.
*/
async calculateSyncTiming(
text: string,
voice: string,
ttsEngine?: string,
geminiApiKey?: string
): Promise<SyncTiming> {
const cleanText = this.cleanText(text);
export function estimateVoiceTiming(text: string): SyncTiming {
const words = text
.replace(/#[\wÀ-ÿ]+/g, '')
.replace(/https?:\/\/\S+/g, '')
.split(/\s+/)
.filter((w) => /[A-Za-z0-9À-ÿ]/.test(w));
const pauses = (text.match(/[.!?;:]/g) ?? []).length;
const punctuationPause = 0.35;
const audioDuration = words.length / WORDS_PER_SECOND + pauses * punctuationPause;
// 1. Générer le TTS preview (avec la même voix que le rendu) et mesurer sa durée exacte
const ttsResult = await ffmpegService.previewTTS(cleanText, voice, ttsEngine, geminiApiKey);
if (!ttsResult.success || !ttsResult.audioBase64) {
throw new Error('TTS preview failed: ' + (ttsResult.error || 'unknown'));
}
const audioBuffer = Buffer.from(ttsResult.audioBase64, 'base64');
const metadata = await musicMetadata.parseBuffer(audioBuffer, 'audio/mpeg');
const audioDuration = metadata.format.duration || 0;
// 2. Analyser le texte (compte les mots réellement lus par la voix)
const wordCount = this.calculateWordCount(cleanText);
// 3. Calculer le word_duration
const wordDuration = wordCount > 0 ? audioDuration / wordCount : 0.6;
// 4. Validation
const warnings: string[] = [];
const isHealthy = wordDuration >= 0.25 && wordDuration <= 1.5;
if (wordDuration < 0.25) {
warnings.push('Texte trop long : les mots défileront très vite. Envisagez de raccourcir.');
}
if (wordDuration > 1.2) {
warnings.push('Texte très court : affichage lent.');
}
return {
wordDuration,
audioDuration,
wordCount,
punctuationPause: 0,
isHealthy,
warnings,
};
const warnings: string[] = [];
if (audioDuration > MAX_COMFORTABLE_SECONDS) {
warnings.push(`Texte long : environ ${Math.round(audioDuration)} s de voix. Visez moins de ${MAX_COMFORTABLE_SECONDS} s.`);
}
if (words.length > 0 && words.length < 4) {
warnings.push('Texte très court : la voix ne durera que quelques secondes.');
}
private cleanText(text: string): string {
return text
.replace(/#\w+/g, '')
.replace(/https?:\/\/\S+/g, '')
.replace(/[\uD83C-\uD83E][\uDC00-\uDFFF]|[☀-⛿✀-➿]/g, '')
.trim();
}
private calculateWordCount(text: string): number {
const tokens = text.split(/\s+/).filter(w => w.length > 0);
return tokens.filter(w => /[a-zA-Z0-9À-ſ]/.test(w)).length;
}
return {
wordDuration: words.length ? audioDuration / words.length : 0,
audioDuration,
wordCount: words.length,
punctuationPause,
isHealthy: warnings.length === 0,
warnings,
};
}
export const ttsSyncService = new TtsSyncService();
+16 -3
View File
@@ -5,6 +5,7 @@
*/
import { z } from "zod";
import { TTS_ENGINES, TTS_STYLES } from "./voices";
const optionalText = z
.string()
@@ -24,9 +25,11 @@ export const videoReelParamsSchema = z.object({
description: optionalText,
ttsEnabled: z.boolean().default(false),
ttsVoice: optionalText,
ttsEngine: z.enum(["gemini", "edge"]).optional(),
ttsEngine: z.enum(TTS_ENGINES).optional(),
ttsStyle: z.enum(TTS_STYLES).optional(),
scheduledFor: optionalText,
wordDuration: z.number().positive().max(5).default(0.6),
// Ancien réglage, ignoré : le minutage vient désormais de la voix elle-même
wordDuration: z.number().optional(),
fontSize: z.number().int().min(16).max(200).default(64),
musicVolume: z.number().min(0).max(2).default(0.25),
drawText: z.boolean().default(true),
@@ -45,13 +48,23 @@ export const imagesReelParamsSchema = z.object({
overlayText: optionalText,
musicUrl: optionalText,
musicVolume: z.number().min(0).max(2).default(0.3),
ttsEngine: z.enum(["gemini", "edge"]).optional(),
ttsEnabled: z.boolean().default(true),
ttsEngine: z.enum(TTS_ENGINES).optional(),
ttsVoice: optionalText,
ttsStyle: z.enum(TTS_STYLES).optional(),
storeName: z.string().optional(),
// Fichiers temporaires à supprimer une fois le rendu terminé
tempFiles: z.array(z.string()).default([]),
});
/** Aperçu de la voix. */
export const ttsPreviewSchema = z.object({
text: z.string({ required_error: "Texte requis" }).trim().min(1, "Texte requis").max(2000),
ttsVoice: optionalText,
ttsEngine: z.enum(TTS_ENGINES).optional(),
ttsStyle: z.enum(TTS_STYLES).optional(),
});
export type ImagesReelParams = z.infer<typeof imagesReelParamsSchema>;
export type ReelJobKind = "video" | "images";
+76
View File
@@ -0,0 +1,76 @@
/**
* Voix et styles de lecture proposés pour les Reels.
* Même catalogue que le service Python (ffmpeg-service/app/voices.py).
*/
export const TTS_ENGINES = ["gemini", "edge"] as const;
export type TtsEngine = (typeof TTS_ENGINES)[number];
export const TTS_STYLES = ["neutral", "dynamic", "warm", "calm", "promo"] as const;
export type TtsStyle = (typeof TTS_STYLES)[number];
export interface VoiceOption {
id: string;
label: string;
gender: "female" | "male";
}
export const TTS_STYLE_OPTIONS: { id: TtsStyle; label: string; description: string }[] = [
{ id: "dynamic", label: "Dynamique", description: "Enthousiaste, rythme entraînant" },
{ id: "warm", label: "Chaleureux", description: "Souriant et proche" },
{ id: "promo", label: "Promo", description: "Annonce énergique des offres" },
{ id: "calm", label: "Calme", description: "Posé et rassurant" },
{ id: "neutral", label: "Neutre", description: "Lecture sans consigne" },
];
export const GEMINI_VOICES: VoiceOption[] = [
{ id: "Kore", label: "Kore — ferme", gender: "female" },
{ id: "Aoede", label: "Aoede — légère", gender: "female" },
{ id: "Leda", label: "Leda — jeune", gender: "female" },
{ id: "Zephyr", label: "Zephyr — lumineuse", gender: "female" },
{ id: "Callirrhoe", label: "Callirrhoe — décontractée", gender: "female" },
{ id: "Autonoe", label: "Autonoe — lumineuse", gender: "female" },
{ id: "Despina", label: "Despina — douce", gender: "female" },
{ id: "Erinome", label: "Erinome — claire", gender: "female" },
{ id: "Laomedeia", label: "Laomedeia — enjouée", gender: "female" },
{ id: "Achernar", label: "Achernar — tendre", gender: "female" },
{ id: "Gacrux", label: "Gacrux — mûre", gender: "female" },
{ id: "Pulcherrima", label: "Pulcherrima — affirmée", gender: "female" },
{ id: "Vindemiatrix", label: "Vindemiatrix — délicate", gender: "female" },
{ id: "Sulafat", label: "Sulafat — chaleureuse", gender: "female" },
{ id: "Charon", label: "Charon — informative", gender: "male" },
{ id: "Puck", label: "Puck — enjouée", gender: "male" },
{ id: "Fenrir", label: "Fenrir — enthousiaste", gender: "male" },
{ id: "Orus", label: "Orus — ferme", gender: "male" },
{ id: "Enceladus", label: "Enceladus — soufflée", gender: "male" },
{ id: "Iapetus", label: "Iapetus — claire", gender: "male" },
{ id: "Umbriel", label: "Umbriel — décontractée", gender: "male" },
{ id: "Algieba", label: "Algieba — veloutée", gender: "male" },
{ id: "Algenib", label: "Algenib — rocailleuse", gender: "male" },
{ id: "Rasalgethi", label: "Rasalgethi — informative", gender: "male" },
{ id: "Alnilam", label: "Alnilam — ferme", gender: "male" },
{ id: "Schedar", label: "Schedar — posée", gender: "male" },
{ id: "Achird", label: "Achird — amicale", gender: "male" },
{ id: "Zubenelgenubi", label: "Zubenelgenubi — naturelle", gender: "male" },
{ id: "Sadachbia", label: "Sadachbia — vive", gender: "male" },
{ id: "Sadaltager", label: "Sadaltager — experte", gender: "male" },
];
export const EDGE_VOICES: VoiceOption[] = [
{ id: "fr-FR-VivienneMultilingualNeural", label: "Vivienne", gender: "female" },
{ id: "fr-FR-DeniseNeural", label: "Denise", gender: "female" },
{ id: "fr-FR-EloiseNeural", label: "Eloise", gender: "female" },
{ id: "fr-FR-RemyMultilingualNeural", label: "Rémy", gender: "male" },
{ id: "fr-FR-HenriNeural", label: "Henri", gender: "male" },
];
export const DEFAULT_VOICE: Record<TtsEngine, string> = {
gemini: "Kore",
edge: "fr-FR-VivienneMultilingualNeural",
};
export const DEFAULT_TTS_STYLE: TtsStyle = "dynamic";
export function voicesFor(engine: TtsEngine): VoiceOption[] {
return engine === "gemini" ? GEMINI_VOICES : EDGE_VOICES;
}