From 3c6a66bf64e003f84e6fc09dcc025f69ec4488a2 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 24 Sep 2026 12:31:27 +0000 Subject: [PATCH] =?UTF-8?q?feat(reels):=20voix=20Gemini=20compl=C3=A8te,?= =?UTF-8?q?=20sous-titres=20mot=20=C3=A0=20mot=20et=20rendu=20de=20qualit?= =?UTF-8?q?=C3=A9?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Service ffmpeg-api réécrit (ffmpeg-service/app/) : - Appels FFmpeg asynchrones : le service ne se fige plus pendant un rendu - Vidéo récupérée par téléchargement (GET /files/…) au lieu de base64 en JSON - Vraies erreurs HTTP ; échec explicite si la voix demandée est impossible - 30 voix Gemini + ton de lecture (dynamique, chaleureux, promo, calme), clé en en-tête, modèle configurable avec repli - Edge TTS 7 (minutage des mots restauré), secours en voix françaises - Voix traitée : filtre, compression, niveau constant ; musique bouclée et baissée automatiquement sous la voix ; mix final à -14 LUFS - Sous-titres calés mot à mot (Whisper pour Gemini et la voix d'origine, à la place de ffsubsync), style Montserrat, placés hors des boutons Reels - Vidéo : plus de retouche luminosité forcée, scaling lanczos, HDR iPhone converti, BT.709, AAC 48 kHz 192k ; la vidéo s'allonge si la voix dépasse - Grand logo de fin affiché après la voix ; FFmpeg 7.0.2 épinglé, polices et modèle Whisper intégrés à l'image ; tests pytest et ruff Application : - Sélecteur de voix partagé (4 pages) : moteur, 30 voix, ton, écoute - Reel images : minutage réel des mots, interrupteur voix respecté - sync-info ne génère plus de voix à chaque frappe (estimation locale) - Stabilisation désactivée par défaut, route /reels/preview inutilisée retirée - Log « [ReelQueue] Worker démarré » pour vérifier la version déployée Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018Ze4bs7tpF1KGWUk6ZZSZ4 --- .env.example | 19 + DOCKER.md | 14 + client/src/components/reels/voice-picker.tsx | 185 +++ client/src/pages/mobile/new-reel.tsx | 110 +- client/src/pages/mobile/remotion-video.tsx | 76 +- client/src/pages/new-reel.tsx | 119 +- client/src/pages/remotion-video.tsx | 76 +- docker-compose.yml | 4 + ffmpeg-service/.dockerignore | 4 + ffmpeg-service/.gitignore | 3 + ffmpeg-service/Dockerfile | 45 +- .../__pycache__/main.cpython-312.pyc | Bin 38831 -> 0 bytes ffmpeg-service/app/__init__.py | 1 + ffmpeg-service/app/align.py | 132 ++ ffmpeg-service/app/api.py | 304 ++++ ffmpeg-service/app/audio.py | 60 + ffmpeg-service/app/config.py | 32 + ffmpeg-service/app/jobs.py | 44 + ffmpeg-service/app/proc.py | 78 + ffmpeg-service/app/render.py | 217 +++ ffmpeg-service/app/subtitles.py | 128 ++ ffmpeg-service/app/text.py | 20 + ffmpeg-service/app/tts/__init__.py | 63 + ffmpeg-service/app/tts/edge.py | 58 + ffmpeg-service/app/tts/gemini.py | 102 ++ ffmpeg-service/app/voices.py | 131 ++ ffmpeg-service/main.py | 1390 +---------------- ffmpeg-service/requirements-dev.txt | 3 + ffmpeg-service/requirements.txt | 16 +- ffmpeg-service/ruff.toml | 9 + ffmpeg-service/test.ass | 14 - ffmpeg-service/test2.ass | 13 - ffmpeg-service/tests/__init__.py | 0 ffmpeg-service/tests/conftest.py | 3 + ffmpeg-service/tests/test_align.py | 31 + ffmpeg-service/tests/test_render.py | 77 + ffmpeg-service/tests/test_subtitles.py | 23 + ffmpeg-service/tests/test_voices.py | 23 + server/routes/reels.ts | 134 +- server/routes/remotion.ts | 2 + server/services/ffmpeg.ts | 433 ++--- server/services/reels/imagesPipeline.ts | 86 +- server/services/reels/queue.ts | 1 + server/services/reels/videoPipeline.ts | 16 +- server/services/ttsSync.ts | 97 +- shared/reel.ts | 19 +- shared/voices.ts | 76 + 47 files changed, 2206 insertions(+), 2285 deletions(-) create mode 100644 client/src/components/reels/voice-picker.tsx create mode 100644 ffmpeg-service/.dockerignore create mode 100644 ffmpeg-service/.gitignore delete mode 100644 ffmpeg-service/__pycache__/main.cpython-312.pyc create mode 100644 ffmpeg-service/app/__init__.py create mode 100644 ffmpeg-service/app/align.py create mode 100644 ffmpeg-service/app/api.py create mode 100644 ffmpeg-service/app/audio.py create mode 100644 ffmpeg-service/app/config.py create mode 100644 ffmpeg-service/app/jobs.py create mode 100644 ffmpeg-service/app/proc.py create mode 100644 ffmpeg-service/app/render.py create mode 100644 ffmpeg-service/app/subtitles.py create mode 100644 ffmpeg-service/app/text.py create mode 100644 ffmpeg-service/app/tts/__init__.py create mode 100644 ffmpeg-service/app/tts/edge.py create mode 100644 ffmpeg-service/app/tts/gemini.py create mode 100644 ffmpeg-service/app/voices.py create mode 100644 ffmpeg-service/requirements-dev.txt create mode 100644 ffmpeg-service/ruff.toml delete mode 100644 ffmpeg-service/test.ass delete mode 100644 ffmpeg-service/test2.ass create mode 100644 ffmpeg-service/tests/__init__.py create mode 100644 ffmpeg-service/tests/conftest.py create mode 100644 ffmpeg-service/tests/test_align.py create mode 100644 ffmpeg-service/tests/test_render.py create mode 100644 ffmpeg-service/tests/test_subtitles.py create mode 100644 ffmpeg-service/tests/test_voices.py create mode 100644 shared/voices.ts diff --git a/.env.example b/.env.example index 4fede88..7dffdb9 100644 --- a/.env.example +++ b/.env.example @@ -105,3 +105,22 @@ LEGAL_CONTACT_EMAIL= # Clé d'accès pour l'API externe /api/v1/* # Générer avec: openssl rand -hex 32 EXTERNAL_API_KEY=your-secure-random-key-here + +# ========================================== +# REELS : VOIX ET RENDU VIDÉO (service ffmpeg-api) +# ========================================== +# Clé partagée entre l'application et le service FFmpeg (CHANGEZ CETTE VALEUR). +FFMPEG_API_KEY=change-me-ffmpeg-key +# +# Clé Google AI Studio pour les voix Gemini (sinon : voix Edge, gratuite). +# Peut aussi être saisie dans l'application, ce qui est prioritaire. +GEMINI_API_KEY= +# +# Modèle Gemini TTS (défaut : gemini-2.5-flash-preview-tts ; repli automatique +# sur celui-ci si le modèle choisi n'existe pas). +# GEMINI_TTS_MODEL= +# +# Modèle Whisper utilisé pour caler les sous-titres mot à mot sur la voix +# (tiny, base, small). Choisi à la construction de l'image : changer la valeur +# demande de reconstruire ffmpeg-api. +# WHISPER_MODEL=base diff --git a/DOCKER.md b/DOCKER.md index c7c6e15..5584015 100644 --- a/DOCKER.md +++ b/DOCKER.md @@ -174,6 +174,20 @@ La stack se construit depuis les sources : `app` et `ffmpeg-api` ont une section `socialflow-ffmpeg-api:latest`) : il ne dépend donc plus du nom donné à la stack dans Portainer. +## 🎙️ Service ffmpeg-api (voix et rendu des Reels) + +- **Premier build plus long** : l'image embarque FFmpeg 7.0.2 (version + épinglée), les polices des sous-titres et le modèle Whisper qui cale les + sous-titres mot à mot sur la voix (~150 Mo avec `WHISPER_MODEL=base`). +- **Voix Gemini** : renseignez `GEMINI_API_KEY` (ou la clé dans l'application). + Sans clé, la voix Edge gratuite est utilisée et un avertissement apparaît + dans les logs. +- **Vérifier la version déployée** : au démarrage, les logs de `socialflow-app` + affichent `[ReelQueue] Worker démarré`, et `socialflow-ffmpeg` répond + `{"status":"ok","version":2}` sur `/health`. +- **Tests du service** : `pip install -r requirements-dev.txt`, puis `pytest` + et `ruff check .` dans `ffmpeg-service/`. + ## 🔒 Sécurité en production 1. **Variables d'environnement** : Ne commitez JAMAIS le fichier `.env` diff --git a/client/src/components/reels/voice-picker.tsx b/client/src/components/reels/voice-picker.tsx new file mode 100644 index 0000000..4540054 --- /dev/null +++ b/client/src/components/reels/voice-picker.tsx @@ -0,0 +1,185 @@ +import { useEffect, useRef, useState } from "react"; +import { Loader2, Play, Square } from "lucide-react"; +import { Button } from "@/components/ui/button"; +import { Label } from "@/components/ui/label"; +import { + Select, + SelectContent, + SelectGroup, + SelectItem, + SelectLabel, + SelectTrigger, + SelectValue, +} from "@/components/ui/select"; +import { useToast } from "@/hooks/use-toast"; +import { apiRequest, getErrorMessage } from "@/lib/queryClient"; +import { + DEFAULT_VOICE, + TTS_STYLE_OPTIONS, + voicesFor, + type TtsEngine, + type TtsStyle, +} from "@shared/voices"; + +export interface VoiceSettings { + engine: TtsEngine; + voice: string; + style: TtsStyle; +} + +interface VoicePickerProps { + value: VoiceSettings; + onChange: (value: VoiceSettings) => void; + /** Texte lu par le bouton « Tester la voix ». */ + sampleText?: string; + compact?: boolean; +} + +const FALLBACK_SAMPLE = "Découvrez nos nouveautés en magasin, on vous attend !"; + +/** + * Choix du moteur, de la voix et du ton, avec écoute de l'aperçu. + * Partagé par les pages Reel (vidéo et images, bureau et mobile). + */ +export function VoicePicker({ value, onChange, sampleText, compact = false }: VoicePickerProps) { + const { toast } = useToast(); + const [loading, setLoading] = useState(false); + const [playing, setPlaying] = useState(false); + const audioRef = useRef(null); + + useEffect(() => () => audioRef.current?.pause(), []); + + const voices = voicesFor(value.engine); + const groups = [ + { label: "Voix féminines", items: voices.filter((v) => v.gender === "female") }, + { label: "Voix masculines", items: voices.filter((v) => v.gender === "male") }, + ]; + + const setEngine = (engine: TtsEngine) => { + if (engine !== value.engine) onChange({ ...value, engine, voice: DEFAULT_VOICE[engine] }); + }; + + const stop = () => { + audioRef.current?.pause(); + setPlaying(false); + }; + + const preview = async () => { + if (playing) return stop(); + setLoading(true); + try { + const response = await apiRequest("POST", "/api/reels/tts-preview", { + text: sampleText?.trim() || FALLBACK_SAMPLE, + ttsEngine: value.engine, + ttsVoice: value.voice, + ttsStyle: value.style, + }); + const data = await response.json(); + for (const warning of data.warnings ?? []) { + toast({ title: "Voix de secours utilisée", description: warning }); + } + const audio = new Audio(`data:audio/mpeg;base64,${data.audioBase64}`); + audioRef.current?.pause(); + audioRef.current = audio; + audio.onended = () => setPlaying(false); + await audio.play(); + setPlaying(true); + } catch (error) { + toast({ + title: "Impossible de tester la voix", + description: getErrorMessage(error, "La voix n'a pas pu être générée."), + variant: "destructive", + }); + } finally { + setLoading(false); + } + }; + + const labelClass = compact ? "text-xs font-medium" : "text-sm font-medium"; + + return ( +
+
+ +
+ + +
+
+ +
+ + +
+ +
+ + +
+ + +
+ ); +} diff --git a/client/src/pages/mobile/new-reel.tsx b/client/src/pages/mobile/new-reel.tsx index 20e39c2..c90bf50 100644 --- a/client/src/pages/mobile/new-reel.tsx +++ b/client/src/pages/mobile/new-reel.tsx @@ -16,6 +16,8 @@ import { Switch } from "@/components/ui/switch"; import { Label } from "@/components/ui/label"; import { Slider } from "@/components/ui/slider"; import { useToast } from "@/hooks/use-toast"; +import { VoicePicker, type VoiceSettings } from "@/components/reels/voice-picker"; +import { DEFAULT_TTS_STYLE, DEFAULT_VOICE } from "@shared/voices"; import { apiRequest, queryClient, handleUnauthorized } from "@/lib/queryClient"; import type { SocialPage, Media } from "@shared/schema"; import { SiFacebook, SiTiktok } from "react-icons/si"; @@ -59,27 +61,17 @@ export default function MobileNewReel() { // État TTS const [ttsEnabled, setTtsEnabled] = useState(false); - const [ttsEngine, setTtsEngine] = useState<'edge' | 'gemini'>('edge'); - const [ttsVoice, setTtsVoice] = useState('fr-FR-VivienneMultilingualNeural'); - - // Gemini native TTS voices (Charon = homme, Kore = femme) - const geminiVoices = [ - { label: 'Charon - Voix Homme', value: 'fr-FR-Standard-B' }, - { label: 'Kore - Voix Femme', value: 'fr-FR-Standard-A' }, - ]; - - // French Edge TTS voices - const edgeVoices = [ - { label: 'Vivienne (Femme)', value: 'fr-FR-VivienneMultilingualNeural' }, - { label: 'Henri (Homme)', value: 'fr-FR-HenriNeural' }, - { label: 'Denise (Femme)', value: 'fr-FR-DeniseNeural' }, - { label: 'Rémy (Homme)', value: 'fr-FR-RemyMultilingualNeural' }, - { label: 'Jenny (Anglaise, Femme)', value: 'en-US-JennyNeural' }, - ]; + const [voiceSettings, setVoiceSettings] = useState({ + engine: 'gemini', + voice: DEFAULT_VOICE.gemini, + style: DEFAULT_TTS_STYLE, + }); + const { engine: ttsEngine, voice: ttsVoice, style: ttsStyle } = voiceSettings; // État audio preview const [isPlaying, setIsPlaying] = useState(null); - const [stabilize, setStabilize] = useState(true); // Activé par défaut pour les Reels + // Désactivée par défaut : double le temps de rendu, utile seulement pour une vidéo tremblée + const [stabilize, setStabilize] = useState(false); const [enableEndingEffect, setEnableEndingEffect] = useState(true); // TTS Sync state @@ -271,6 +263,7 @@ export default function MobileNewReel() { ttsEnabled, ttsEngine, ttsVoice, + ttsStyle, enableEndingEffect, }); }; @@ -492,82 +485,13 @@ export default function MobileNewReel() { TTS — voix activée - {/* TTS Engine Selector */} -
- -
- - -
-
- - {/* Voice Selector */} -
- - -
-
- +
{syncInfo && ( diff --git a/client/src/pages/mobile/remotion-video.tsx b/client/src/pages/mobile/remotion-video.tsx index b6b302d..b27526f 100644 --- a/client/src/pages/mobile/remotion-video.tsx +++ b/client/src/pages/mobile/remotion-video.tsx @@ -1,5 +1,7 @@ import { useState, useRef } from "react"; import { useToast } from "@/hooks/use-toast"; +import { VoicePicker, type VoiceSettings } from "@/components/reels/voice-picker"; +import { DEFAULT_TTS_STYLE, DEFAULT_VOICE } from "@shared/voices"; import { Card, CardContent, CardHeader, CardTitle } from "@/components/ui/card"; import { Button } from "@/components/ui/button"; import { Textarea } from "@/components/ui/textarea"; @@ -33,23 +35,11 @@ export default function MobileRemotionVideoPage() { const [productInfo, setProductInfo] = useState(""); const [generatedVariants, setGeneratedVariants] = useState([]); const [ttsEnabled, setTtsEnabled] = useState(true); - const [ttsEngine, setTtsEngine] = useState<'edge' | 'gemini'>('edge'); - const [ttsVoice, setTtsVoice] = useState('fr-FR-VivienneMultilingualNeural'); - - // Gemini native TTS voices (Charon = homme, Kore = femme) - const geminiVoices = [ - { label: 'Charon - Voix Homme', value: 'fr-FR-Standard-B' }, - { label: 'Kore - Voix Femme', value: 'fr-FR-Standard-A' }, - ]; - - // French Edge TTS voices - const edgeVoices = [ - { label: 'Vivienne (Femme)', value: 'fr-FR-VivienneMultilingualNeural' }, - { label: 'Henri (Homme)', value: 'fr-FR-HenriNeural' }, - { label: 'Denise (Femme)', value: 'fr-FR-DeniseNeural' }, - { label: 'Rémy (Homme)', value: 'fr-FR-RemyMultilingualNeural' }, - { label: 'Jenny (Anglaise, Femme)', value: 'en-US-JennyNeural' }, - ]; + const [voiceSettings, setVoiceSettings] = useState({ + engine: 'gemini', + voice: DEFAULT_VOICE.gemini, + style: DEFAULT_TTS_STYLE, + }); const [musicFile, setMusicFile] = useState(null); const [selectedTrack, setSelectedTrack] = useState(null); const [musicVolume, setMusicVolume] = useState(0.3); @@ -143,16 +133,6 @@ export default function MobileRemotionVideoPage() { const totalSelected = images.length + selectedLibraryImages.length; - const handleTtsPreview = async () => { - const ttsText = stripForTTS(overlayText); - if (!ttsText) return; - try { - const r = await apiRequest('POST', '/api/reels/tts-preview', { text: ttsText, ttsEngine, ttsVoice }); - const data = await r.json(); - if (data.success && data.audioBase64) new window.Audio(`data:audio/mp3;base64,${data.audioBase64}`).play(); - } catch { toast({ title: "Erreur prévisualisation voix", variant: "destructive" }); } - }; - const togglePlayPreview = (track: AudioTrack) => { if (!audioRef.current) return; if (isPlaying === track.id) { audioRef.current.pause(); setIsPlaying(null); } @@ -168,8 +148,10 @@ export default function MobileRemotionVideoPage() { images.forEach(img => formData.append("images", img)); selectedLibraryImages.forEach(m => formData.append("existingImageUrls", m.originalUrl)); if (overlayText) formData.append("overlayText", overlayText); - formData.append("ttsEngine", ttsEngine); - formData.append("ttsVoice", ttsVoice); + formData.append("ttsEnabled", String(ttsEnabled)); + formData.append("ttsEngine", voiceSettings.engine); + formData.append("ttsVoice", voiceSettings.voice); + formData.append("ttsStyle", voiceSettings.style); if (selectedPageIds[0]) formData.append("selectedPageId", selectedPageIds[0]); if (musicFile) { formData.append("music", musicFile); formData.append("musicVolume", String(musicVolume)); } else if (selectedTrack) { formData.append("musicTrackUrl", selectedTrack.url); formData.append("musicVolume", String(musicVolume)); } @@ -354,41 +336,7 @@ export default function MobileRemotionVideoPage() {

TTS — voix activée

- {/* TTS Engine Selector */} -
- -
- - -
-
- - {/* Voice Selector */} -
- - -
- - +
)} diff --git a/client/src/pages/new-reel.tsx b/client/src/pages/new-reel.tsx index 64717b3..09a5c1c 100644 --- a/client/src/pages/new-reel.tsx +++ b/client/src/pages/new-reel.tsx @@ -25,6 +25,8 @@ import { SelectValue, } from "@/components/ui/select"; import { useToast } from "@/hooks/use-toast"; +import { VoicePicker, type VoiceSettings } from "@/components/reels/voice-picker"; +import { DEFAULT_TTS_STYLE, DEFAULT_VOICE } from "@shared/voices"; import { apiRequest, queryClient, handleUnauthorized, getErrorMessage } from "@/lib/queryClient"; import type { SocialPage, Media } from "@shared/schema"; import { SiFacebook, SiTiktok } from "react-icons/si"; @@ -67,7 +69,8 @@ export default function NewReel() { const [musicVolume, setMusicVolume] = useState([25]); const [ttsEnabled, setTtsEnabled] = useState(true); const [drawText, setDrawText] = useState(true); - const [stabilize, setStabilize] = useState(true); // default to true + // Désactivée par défaut : double le temps de rendu, utile seulement pour une vidéo tremblée + const [stabilize, setStabilize] = useState(false); const [enableEndingEffect, setEnableEndingEffect] = useState(true); // TTS Sync state @@ -79,24 +82,13 @@ export default function NewReel() { warnings: string[]; } | null>(null); - // TTS Engine & Voice - const [ttsEngine, setTtsEngine] = useState<'edge' | 'gemini'>('edge'); - const [ttsVoice, setTtsVoice] = useState('fr-FR-VivienneMultilingualNeural'); - - // Gemini native TTS voices (Charon = homme, Kore = femme) - const geminiVoices = [ - { label: 'Charon - Voix Homme', value: 'fr-FR-Standard-B' }, - { label: 'Kore - Voix Femme', value: 'fr-FR-Standard-A' }, - ]; - - // French Edge TTS voices - const edgeVoices = [ - { label: 'Vivienne (Femme)', value: 'fr-FR-VivienneMultilingualNeural' }, - { label: 'Henri (Homme)', value: 'fr-FR-HenriNeural' }, - { label: 'Denise (Femme)', value: 'fr-FR-DeniseNeural' }, - { label: 'Rémy (Homme)', value: 'fr-FR-RemyMultilingualNeural' }, - { label: 'Jenny (Anglaise, Femme)', value: 'en-US-JennyNeural' }, - ]; + // Voix : moteur, voix et ton de lecture + const [voiceSettings, setVoiceSettings] = useState({ + engine: 'gemini', + voice: DEFAULT_VOICE.gemini, + style: DEFAULT_TTS_STYLE, + }); + const { engine: ttsEngine, voice: ttsVoice, style: ttsStyle } = voiceSettings; // Enable TTS by default on mobile useEffect(() => { @@ -413,6 +405,7 @@ export default function NewReel() { ttsEnabled, ttsEngine, ttsVoice, + ttsStyle, drawText, stabilize: stabilize, enableEndingEffect, @@ -819,88 +812,12 @@ export default function NewReel() { TTS — voix activée - {/* TTS Engine Selector */} -
- -
- - -
-
- - {/* Voice Selector */} -
- - -
- -
- +
+
{syncInfo && ( diff --git a/client/src/pages/remotion-video.tsx b/client/src/pages/remotion-video.tsx index 7c3e4c5..111b1fb 100644 --- a/client/src/pages/remotion-video.tsx +++ b/client/src/pages/remotion-video.tsx @@ -2,6 +2,8 @@ import { useState, useRef } from "react"; import Sidebar from "@/components/sidebar"; import TopBar from "@/components/topbar"; import { useToast } from "@/hooks/use-toast"; +import { VoicePicker, type VoiceSettings } from "@/components/reels/voice-picker"; +import { DEFAULT_TTS_STYLE, DEFAULT_VOICE } from "@shared/voices"; import { Card, CardContent, CardDescription, CardHeader, CardTitle } from "@/components/ui/card"; import { Button } from "@/components/ui/button"; import { Textarea } from "@/components/ui/textarea"; @@ -34,23 +36,11 @@ export default function RemotionVideoPage() { const [productInfo, setProductInfo] = useState(""); const [generatedVariants, setGeneratedVariants] = useState([]); const [ttsEnabled, setTtsEnabled] = useState(true); - const [ttsEngine, setTtsEngine] = useState<'edge' | 'gemini'>('edge'); - const [ttsVoice, setTtsVoice] = useState('fr-FR-VivienneMultilingualNeural'); - - // Gemini native TTS voices (Charon = homme, Kore = femme) - const geminiVoices = [ - { label: 'Charon - Voix Homme', value: 'fr-FR-Standard-B' }, - { label: 'Kore - Voix Femme', value: 'fr-FR-Standard-A' }, - ]; - - // French Edge TTS voices - const edgeVoices = [ - { label: 'Vivienne (Femme)', value: 'fr-FR-VivienneMultilingualNeural' }, - { label: 'Henri (Homme)', value: 'fr-FR-HenriNeural' }, - { label: 'Denise (Femme)', value: 'fr-FR-DeniseNeural' }, - { label: 'Rémy (Homme)', value: 'fr-FR-RemyMultilingualNeural' }, - { label: 'Jenny (Anglaise, Femme)', value: 'en-US-JennyNeural' }, - ]; + const [voiceSettings, setVoiceSettings] = useState({ + engine: 'gemini', + voice: DEFAULT_VOICE.gemini, + style: DEFAULT_TTS_STYLE, + }); const [musicFile, setMusicFile] = useState(null); const [selectedTrack, setSelectedTrack] = useState(null); const [musicVolume, setMusicVolume] = useState(0.3); @@ -109,16 +99,6 @@ export default function RemotionVideoPage() { }); }; - const handleTtsPreview = async () => { - const ttsText = overlayText.replace(/#\w+/g, '').replace(/[\uD800-\uDFFF\u2600-\u27BF]/g, '').replace(/\s+/g, ' ').trim(); - if (!ttsText) return; - try { - const r = await apiRequest('POST', '/api/reels/tts-preview', { text: ttsText, ttsEngine, ttsVoice }); - const data = await r.json(); - if (data.success && data.audioBase64) new window.Audio(`data:audio/mp3;base64,${data.audioBase64}`).play(); - } catch { toast({ title: "Erreur prévisualisation voix", variant: "destructive" }); } - }; - const togglePlayPreview = (track: AudioTrack) => { if (!audioRef.current) return; if (isPlaying === track.id) { audioRef.current.pause(); setIsPlaying(null); } @@ -134,8 +114,10 @@ export default function RemotionVideoPage() { images.forEach(img => formData.append("images", img)); selectedLibraryImages.forEach(m => formData.append("existingImageUrls", m.originalUrl)); if (overlayText) formData.append("overlayText", overlayText); - formData.append("ttsEngine", ttsEngine); - formData.append("ttsVoice", ttsVoice); + formData.append("ttsEnabled", String(ttsEnabled)); + formData.append("ttsEngine", voiceSettings.engine); + formData.append("ttsVoice", voiceSettings.voice); + formData.append("ttsStyle", voiceSettings.style); if (selectedPageIds[0]) formData.append("selectedPageId", selectedPageIds[0]); if (musicFile) { formData.append("music", musicFile); formData.append("musicVolume", String(musicVolume)); } else if (selectedTrack) { formData.append("musicTrackUrl", selectedTrack.url); formData.append("musicVolume", String(musicVolume)); } @@ -294,41 +276,7 @@ export default function RemotionVideoPage() {

TTS — voix activée

- {/* TTS Engine Selector */} -
- -
- - -
-
- - {/* Voice Selector */} -
- - -
- - +
)} diff --git a/docker-compose.yml b/docker-compose.yml index a30659d..13dbb71 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -71,12 +71,16 @@ services: build: context: ./ffmpeg-service dockerfile: Dockerfile + args: + # Modèle Whisper (calage des sous-titres) intégré à l'image + WHISPER_MODEL: ${WHISPER_MODEL:-base} image: socialflow-ffmpeg-api:latest pull_policy: build container_name: socialflow-ffmpeg restart: unless-stopped environment: API_KEY: ${FFMPEG_API_KEY:-socialflow-secret-ffmpeg-key} + GEMINI_TTS_MODEL: ${GEMINI_TTS_MODEL:-gemini-2.5-flash-preview-tts} networks: - internal expose: diff --git a/ffmpeg-service/.dockerignore b/ffmpeg-service/.dockerignore new file mode 100644 index 0000000..665490f --- /dev/null +++ b/ffmpeg-service/.dockerignore @@ -0,0 +1,4 @@ +__pycache__ +.pytest_cache +.ruff_cache +tests diff --git a/ffmpeg-service/.gitignore b/ffmpeg-service/.gitignore new file mode 100644 index 0000000..b267296 --- /dev/null +++ b/ffmpeg-service/.gitignore @@ -0,0 +1,3 @@ +__pycache__/ +.pytest_cache/ +.ruff_cache/ diff --git a/ffmpeg-service/Dockerfile b/ffmpeg-service/Dockerfile index 0a648fe..6e466c4 100644 --- a/ffmpeg-service/Dockerfile +++ b/ffmpeg-service/Dockerfile @@ -1,38 +1,39 @@ FROM python:3.11-slim -# Install system dependencies including FFmpeg with vidstab support and fonts -# We need to build/install libvidstab and compile ffmpeg with it, OR use static build -RUN apt-get update && apt-get install -y \ - wget \ - xz-utils \ - fonts-dejavu \ - fontconfig \ - build-essential \ - python3-dev \ +# Polices des sous-titres (Montserrat) et des emojis, installées au build +# plutôt que téléchargées à chaque démarrage. +RUN apt-get update && apt-get install -y --no-install-recommends \ + wget xz-utils ca-certificates fontconfig \ + fonts-montserrat fonts-dejavu-core fonts-noto-color-emoji \ && rm -rf /var/lib/apt/lists/* -# Download static FFmpeg build with all filters including vidstab -RUN wget -q https://johnvansickle.com/ffmpeg/releases/ffmpeg-release-amd64-static.tar.xz \ - && tar xf ffmpeg-release-amd64-static.tar.xz \ - && mv ffmpeg-*-amd64-static/ffmpeg /usr/local/bin/ \ - && mv ffmpeg-*-amd64-static/ffprobe /usr/local/bin/ \ +# FFmpeg statique (vidstab, zimg pour le tone-mapping HDR, libass). +# Version épinglée : une mise à jour ne change plus le rendu à notre insu. +ARG FFMPEG_VERSION=7.0.2 +RUN wget -q https://johnvansickle.com/ffmpeg/releases/ffmpeg-${FFMPEG_VERSION}-amd64-static.tar.xz \ + || wget -q https://johnvansickle.com/ffmpeg/old-releases/ffmpeg-${FFMPEG_VERSION}-amd64-static.tar.xz \ + && tar xf ffmpeg-${FFMPEG_VERSION}-amd64-static.tar.xz \ + && mv ffmpeg-*-amd64-static/ffmpeg ffmpeg-*-amd64-static/ffprobe /usr/local/bin/ \ && rm -rf ffmpeg-* \ - && fc-cache -fv + && fc-cache -f WORKDIR /app -# Copy requirements first for better caching COPY requirements.txt . RUN pip install --no-cache-dir -r requirements.txt -# Copy application code +# Modèle Whisper (calage des sous-titres) téléchargé au build : aucun accès +# réseau nécessaire au premier rendu. +ARG WHISPER_MODEL=base +ENV WHISPER_MODEL=${WHISPER_MODEL} \ + HF_HOME=/opt/hf-cache \ + HOME=/tmp \ + XDG_CACHE_HOME=/tmp/.cache +RUN python -c "from faster_whisper import WhisperModel; WhisperModel('${WHISPER_MODEL}', device='cpu', compute_type='int8')" + COPY main.py . +COPY app ./app -# Create API Key env var (should be overridden in docker-compose) -ENV API_KEY=default-dev-key - -# Expose port EXPOSE 8000 -# Run the application CMD ["uvicorn", "main:app", "--host", "0.0.0.0", "--port", "8000"] diff --git a/ffmpeg-service/__pycache__/main.cpython-312.pyc b/ffmpeg-service/__pycache__/main.cpython-312.pyc deleted file mode 100644 index 46bcd9dfdb8378751eaa61f44121577837f32e49..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 38831 zcmdVD3wT@CbuM`DBHk|&B*8cMCis3z6eUWc_>wG%vMu^WOJyJ)P$ETw^Z_V|FlaN0 z+fa#GQ%!C~H})86(wd%V8hUO!VW*w1vJ*E-bMGx6&>;dVbLB=exBk91p)94zRr-B1 z|Jnx!0BOpOdvCwFb4cv7&wfAGUTf{O*Z!f=sN?Yb!|gvAi@e5hKcx@pF~%+T{*!{^ zu5diZ%O}?z}vbSl-gtuzKJZTxS#M7`2S(#t@P&#{O3}vvlZOA6)+_G=Qi#U|& z){WOn1y8=m8Z^^~4Av;pB@S1PQ;@>&$obfg(Wb(~xdF?fw6boh3Z=WA; z*n>#C@gcq@=4*b58_Gkzc~a?RoI7*dl3Jv4@z$3(DfCc2pN(ljTVM<%nN_AHpH8O88gvRrnT5>_iD`SlDW}1vRL_cWsJ)Exzkmz3Wi# z`jmRtKctNtq&7y|DR`yyH7es9(MEE<=_OU7MH`TYi#6jWgKw5%eJqU@w}Yjl+5>-P|SF6aPZOI3nT7n-`JEVmUhtX;@yH%7SkSZdEG~+c=rT8G{=~?YhnN& z%12$kGch${4vtUs&4H#yxfIyyP+K0Q1wOpUm`-Z9VV!8heG z^|VWHdwkxQ#(iPT>l>ar>sM0fnDXGUqrHATi&;P78ad;RnV#U(9Fy;nG2d=6v&$)t9aVT}6XBcJ2Om8sE1J1^%Ir{%noS6)lZk#S0S%QMIE zX#wsWHz(uO>@82pPZN-Ro#V9u`PaGE5+g{KS|VMtM4wU~{b?D>Vo1W8S+11xX{6$m zae{qLk&?GDpg^wjWctqlD$gliP|U{P=Vj-)3yNpB^D?JtV5SraZM^=~uQFZG;h#5L`k5!}O5Grs8=Urgz9 zU+~4W-kDRrG2et6J<1EN^W->FfPDV`)!%mXiLo7{Q!^gE!;uKu>DSjfq|lu+MN+Z4 z&bh`WT&E`7h=DQUs2Uxqo4_b>c05o^R#{Aiu`zOXriL>4!{EAVmg!~J}6@fVFtt9>*GKUHgogbEu}W3Ygkh@ZpihjF4?Ut z#}&_IPt=&X=nol6R$5p2`wDrs?!V|P7b`YYoYuCH#^sa+pAXra7mU&D;^4EP?8XJd zEsZf`sau|09$r0s!*KnDkm<=a%~N;aYHC>1G~Uvrzihf_TGB2)zg+esO~bEl>1`iz zGRlx5?q2O$Z3&t7t!ws&HT!?Pp&*|JUOP(oyIqd1dhYFd?E$mm?VY9rI>kFW73_Xv z1MfaHbGlBf_XvU}9yM%V#P8l8V*VcHWSGgAvx&Jp9$%>w;bGxryc}~@!7HvQCAhZT z+@AZibX1sYpPq_LocZFs=QXd&eF^w5o|u=(@{Q}0cF88KAU)j6@|P7OiZO0PeiHcA z3Dj;*KBpkqaZcuyUD5~S0mWD4m&{6TL@}qBlLutu5*{WchPRX+a3NhT7a%-LpUd%v zSCz++vMe6=JV6eV6b+Cgdrrw4eOAdOkbqTVvJp9h8_E}u0^kSUNx1R9C>ZPzEQlyj z>_H)&tl4B_l92_&DNovC3M>jAHpBJ9`{+u{C9^b_yfc%NE@3vN0Kkc<*d&ix05yiC z0B2}|s@#GwB{;Q0E+wBwMm~&~e99YB$3az0HRf|qdNG-&CjfrNJZ`T*lO?9``UK2% zbv#9(njB?kW3qEXF|1w{44SGAp@e)4Y-|pTV^bjPuaQFi1javbznbSZlx9`&t@QjW z^_T0n%(aZQaDWaHC#p28$(xz*2~+%Zn{+E@uRdcg-4$ZiCZkb^rCAj{R*iY zGHK!#M=jZ}W-LAS^{kND@t(%P=4|o5c#oonUyya{WiNF#9;o8pZZ#Yz(Yzy5k(rhU zmv@TE_nne7vR7%lHHvqd4BZOFJDoDJ?^VI~y9yQT-_f z%FY9*K7++c#$CE9B2!!Pw9lzh%4iL!Hn%K*tA0-Pf@&6hLGL7Rb!PbY_jEYJiK2$hJXF_Op|Q%#=b#{_XX z(+KY8X53y7Op4QPA1#(+UiUCMc-ZT6`DVN^<eM)p zTi5yI8X8mg^gcQ8#E~P;w3vp<17uT(nOUKcAik%GqEpNW6EWoKou2Y|-7(pyxNH(Y zXUnFvZo1j9Wf8psEhhc=d3gZJdF~FOs4Hy*zMnAlSN6wlJ_}*twzOvWr5+ZQ-o8`D3^Bx$FAku)cUr zUrH#bY1f*jV?%D$?w3W=hZnkUWfxruTn+@gSCnhnZ7i{MLwVRxzGkSrWzAfi3i{Tp zwJWk)maKJ4N!U^nY=2|V)jccLHA~Ac61X?lNk zeUWDu*0I{OS{h2R?evc zD!6pxD=m<=9g5DYq!kb4%nZo4L%Tnv_S>xR0YxB9T6?ARv4-?2xl4~Kxqw_KNvUgE z3Ko#ShQL@*1Dntgj6i!OPkIV?(v2*T; zS;v^iu{C2_t1Md`{Ytjt(NdT!fIpMv;;nVe5J$@~Iwp8g9Dy({5#)Lm&$r2w=d z6O8qCIATg_H>XahL!6kBfUd9?R<;ntl;>R&XTcov@ZR%SG-GM5X~1_rrW~Id^TZSr zZVy`*oSK+vdJ0@MFiRNR_6m_Kd zq4fGSLqjwp`^$$n)Chauz^Qa0-P3EzXEx+IReRKuzM#5QT>3`#)$EXS_m7JAgbf7? zJ&R>as`u>$i%P`2(s{WvnDt&pRn(Sq<;dkD-%O{#Z{%OiU*=bOBE`Gjvvu56aXEz> z3a+H{E{Gf}@?U)9Rd2BE>;9#QQ1!m+s_U+|HLLzm`Xg(G!_f@;lJ#=?!ZEN!4Au<; zXGs6meRZ5RYG*u+=9Q+E(vWG#x@Kocv-7sz`s)p~Ox6CtOMI!fwXNOR+}nG*vhezD zwzj)n@m+_hyG8Nc78UH904M=ZdhY!Tj4Pa*8mo|>AP5YE!~1R0Ys z-sJAV%{^(-+QupP~h_W*3D4Ty|QL~hB*%(ol zMe_>b*G4I1W+QcPM43$~A6YCXH+#YS;YC|SSrEizsuWS$SkzQ|1Zh zq{53MCQWdI73~=TqLO%sn5P+d@1J2@0ovk~&C0+`NDUJY48FfiT9~)>S0w$xN0jFE zwtEO*D9FHxqy+#ELMM=%ZBY~zk1|Ol@27^wZe=p`c8H6_L?ATnIOo*(yB`Naf6na{ zi4m+*U{#;-xlY4I1m&4Ao_Bk|CU*&5EJ%Wb9|Hr;HS4ING3}W0Ow2m#W85#O%0K_h zzh>{hpZ^MoFSX!Cz!CQYW$B6YHGahjZ%s_*aHa`!2nv=RQE@S2JkYTCDHq(}$IYCI zsX)1nO}{B)z2~JKiIWjoBNJ|yhfOG~KLSV+Rvh7X5adtr^U~O!=RVMzqh{-Ez4_%m z7xye_f+Z1s`MSO?tgl->vnr41chC3S(OcH_g<*Z+l5hF(HGLzw7+>CTamQliQuC58 zqAyFjR4kP;mvw!0SYI8{*UtAu_2&7*Y?y67sQz!1{M z29bI|m_;bTPsa0P1ju-Sj2B@z)vV8jd2;wX8DAjdi)0MKKyMRIR`_k$Q~Nr7OJ4)5 z3KtRP&nX-Mrg`qi|9$%Tw%+t|`^EOfo-2=Be&nl1BKqQxvY2&lOvM0sOy`>6$EHMz zSL5<}MVlc${MdLLN#zdeQ9P>Py}yO)mt^8rMUom>kYUSM_L7E|_axZ!Qi~GxA*5ZT z%?9OZ#ax<41-pt*BRi;(1aLv>@h6qv=u4EsMBwLEy5|LaP=0AFH{(^uhZDO~zX zWlG}Hr-NHx#5?$6zJxF3%lL91q4Z)|5;XQ{&0L1BSaKgn{#&0whR-3njh9KbQ=h_SBKOKk!_z2wI?Y=u1F@M}{18@Ku|$}hRL)U|LfwXRJRgKuUv zPUVnf&Z(tZBaJU^tvB1)Z1qR`-5xgx2)AQ!@VJ953l5^Z9YkByO}N1`aCm1uBWEBV z0bHL?N-@;ybVNmCyC2vVf7(5a1+Bh&YI1VMGd6+*>qT~Tm6Hi3h{6+&r;zFaqS3Jp z>OkUP3DmM=AdvvFhsNp~sljZ+PS{2HIqnZ+S7d;+Iwd#DQQN#J z8xUp?8gfGN`o@@&s2JhPaPU%M4#&K>fvVke8Vz`1TG;D%l8s7BEip`uNHP=N`aKg< zBd!VW-ugt)5G6yqM!#a4tRZV(xVAS)oPhGL)xr8-nMB1GO;#ClR|2ISi>VHR^)W_b zM>7+Ct4OplCy`3?TSO=Fj=R`4Q+9FJbkViP?LiyI-K=g;-4n;_9svP3D>@A@85V3L zd|4b8K5RBm`^164FbLw~(C$f0#fA)cEiRucW=QScSlZO+sMqZqoBx9)5V;bW%5Y$( zLZtM*jy4cpC1V#Ev<%G9x@CIiBBjbc>^kcla%A zygE*aLu<_K{g5W?4Eh&c<@o#G`F%&FUsXw44i6Od&7*z|q-#*34u@avbz--C6yxud zTiA{80)dqW#yanrsq-L{fw{ZA56r;MF}gaXT0BtHO^rS%iA#NTdZw6c&G31F_*H_p zv#G-~1yQihCWsEP?G;vEwmtLrFTX`XJ5;?c!onOqaRMG~D)Y_|_>6P*x~FC)cn5Lx zCfzQ!OGc5TB6K+Z3xd+K-wE*~A4utm1T!F}g3Ow0cBTv7@ze&Uwzd(rg_uYac!i@b zOghY~Q?8M-G+*PBN}O#@nXrs>X7VYqc+YreMo6q}(~y`U&>7E7tQs-hK#EQKW*s25 zCdZf@+~zp6JjUZt?M6jeCI*7oNJ9^ymrA=PCRmd(H_&b@SiR#2B={UmUIHRcIWdLH zGb&sVG7>?&W0TWh zn;@h!TbvyPZG^9oafyuIr6kp3-r-5tgd0L7=rZaqkv9WTkO{QKFl&pL(Iw7U(T}#p z5@0!I-dY3$o5I5*XJ$NS!7pX+m`YMNXKyA8hutp6W!?x<%WzOY_t4E_1jnS&syJ}@l?k^N+nOPgfT(0x} z1v#g;g|c?7Rd2#J^KSs=;Y@@M_nP;NXX9rLLo<`6mk`~dvVmBAGH@n?Uhlx z6YRIBea|g>-qIsqH*Z+AWm$_y?%Oy=_3JOZ@Z-ro?BWv*>HbyII zmfOF*@6CPdHT%Og`y(|6BIVtS11Mc6f9I;{x;M1*@QtQW?$P&b1HZbJ?SKFha%JA# z*SoH_gmU}WZ3n})gS3wcp9kRJx9q!Gd%b7T{6R(uGRWM)iiXlTqLppY!s36{qxsdXq^J>is*;`jmt{e+x^{iQXe{9LRL+##w!+8D0P}WmF zvON8(J7zoW#HI|{>+>1j}XuR=@q231eH z_Is@dit+LNtghNa3i%J@GJO0%p*>Wl`hmf8s7UpLTot(#smP_QOM#ESt&}6#zjd0( zzLVK^X%9P8Ka^)2&Qtv`GY!5!%u|tzLjk++mnb!34YS2ccoUx^37!EybodeKcJCiy zO1v!4@oD&h&B)7%=;IY+&S{do4KN_%oCQ#4e-am4)XwPw>TT(39j~Mmq&V2A)Fyct z$+UQtALSuh9(;s&obd$f6bZzi1ogTN&x23!Jmw53c_ipSUVZii(l-Y5URA*0RRoOS z8=OLJtd2x;Qb~!;HD`JVCq}y+(>0LDA)pDUD1N{szIk0LM}f7gUXWdS75oz|YK)Mn zZG34NyTD2zH;^XSll+%J8n0)e7!M>V&6RZDgLHg}{pOUKC)z`7U8xVyZ$SI!w5fRn zv?+O{Z7Q9S`IxA*5?>;K5Mp@B`%)FCYcq+S^On~Xb7tNOeh2ZH9!DzplM;j;qiC|!0S1ExJ-gPHT1 z+7BMroaNFW_&kJTvj-fUTt4qwzC^2ni<241+>X-{Fb6EGP2#fw$8Ny$OKuc^%B3b^ z4|o&u@jU4(`A&>x!jrZ_!G%3 z#>f_lFxs6!)+Ch7fB6645&n*X{U;cw|L_%h-rV;AZ4*6vd<44~KF1-?=+sG__%R3? zu)}e#wZ5^jM)xQvj>p~Jr#l=?jqQ!n=QH@+)!c{(KwCll3y?DF^6P0I=kQeQE;a&LPlBV;Kc*ii3@fee7Y@Mm&9e32S2`tnK88r?#%Cwvqic;e&qk*HqhD+g#gL z+tx_`_(m)DGKNdCg^##q-PC^6|5PNjOuSYDti?a6{G+4p5kwjUCa+@~CNbIbFa2# zWBlq%4r$(k?*t|k*iV?h-~?zyPC%{0Zxp9Ne5!y&z-Fb>9Mifz*zG2E3?m2`zau8W2 z5@@H%MwBu`K4NKcipmHc79ILxw4XPY_!Os|Y)mAeh_E9HE@s3$#AzEuKy$cb8kcW) z0!%NQ&td6-CBeti&H&)J1&B8DH}*?TcjBEW5(l4dL{L1@&z^ zNSjx5>vm_@?u^)L*6pofduym|AYwnZV7!%)wKx~aC||aQGpZLfAL!FBPyW#biqrmz zAt)?YUwt8LZ(BGNHKf0M^y1Om`9-fEeC1&9WF)^OT2Q_GM7W?inpYN_3Fp;<2(qT% zSELzDAo&W47LNXXMsC#6up(ROdP@~{>{z)FcI>}w+;DQ~wRamiP1Y+FxAKdEX+hzt zZn-V&Xj<7D$v?1=wjf;8Ew+UP)5WUb0pyZ%D;R>A7|a8v({vT)NtB<~o~c z^vA=R$1kRBXt{JdWv;OY0Y^()0ujrL{rMfGADC{y#A#sPv)*MM59bWvS- z@XQ;NS0}$QwO-K~uILPv?_E7}OK(~@_m$lXb&GS~JPum9d{?NfW3_@rH4n&sp-`yO zeyK(ZA#MEwuK|PWZ;YA!)r#*|tNL5in|2u!82VquG4$<(6C@mBPL;sjNxHEQxQ*r^ zICVYl5SXo6uzq@IUO@*Z zc_r{jV91~2Z?<}g^-4w3WK;4=#leUVEu%V zf;BQ{N=Z+L^i12NX9jj>^rLO&EXloyfIg95%8pb(zZnmlOGk|}DAkiZ^+6KLPHF2z z4+Pk5B~n0Myq?TN9=0S-AF%O>L!iKCHQVj8@vQ z7W%xxL;qE1L zmlk%4G7(?v&z}L4mUz{(?g^YLb+~|w`y69l2SoSJx;?dG%DyvW;ArCzD`9y$TI#UK z$MJPYuMuO~1B`r@alTIjvG?#IN-g|HfGVd_xIt^;56JjiGJZ%#7)BgN6S1`I(6#>& zakNdFVgw4=rr3U1M3D|`i2+BTEip`Ikl3jr=YhnQaeS-H3PEnWv|T1FA`wV95rZ7N zJD2fdSc&js*y3AeaV+DjaPe0p7`jQm_cl9aHwY0SUb?tZmbw4UieTnnmajbawUMRb zE2FFBi=%H=h}&gf$S;(MWYG5!{`B*=*(%2GH!390>L zMa4ZqR!QL=;^M$3t0Jo%ZCgnw)Mu&O8dZ)+aj|kaJKn4bKK5$qlI!(~H*4aR%FLF1 z2)N39NQ@e)#Mx0MnI!y(90}PL*2##HK_wJoWV{a}aR@#YwWdkQ z=3-TR9y%e?{S$=tYvb5;twZ=f@r~m;3zX7vGN`8+B26-GlfqBw(@Vw&WXzEgC zVeH9af5RnUCxnRRI3^zd5CfV-pTq_RByK?!F*<=(JQ$c9cZ5RjE zH&{`m0JOy+V%@aSOMqv)N5s4ToN7YGAJB+^CKC=PZ*w_o<`vy#-B%6shoVMv9QFRG zC1*jIM6n0gG$n+8%|-MDLF-L@NsuS}EYy1Ny$T6CUaDI&R7LX(FCGQDXU)B9;53;) z3|}^0G%ilAY05S%pNNNIi(>u=(7%kVFipQm=Ke_f{`tdEb^4;~raC*CmA#<BDu#F z4&TZxSUS7h9LcR;I1IM_>jsbp$CpbZ_WE`Ej<9{lig&d;V&6|xLe9Fu5jHr2JC-C{W9}gLh-&M%9x`i~9Qj497u%_aoVWWYw7v69F&9P{RZGL<5&B<>~t=IR4>w80W zeb>+Y2XE;3Ge3O(#$@POe$9O*G$usc-VpB#`TQZx^I-IEMWf$s%j{2Ad_P^)pQjdN z7&rn>5OJFjY7K@IP-*TSR0@IR4cS{I7I8U{O%)IEyA+U@fbDbY6xW1E6HsezTLemp zqeBG#@NDa!qHkzU3l=U=bd-RJ-|@PD7Iu9cVMqB@qjf{H}=j8lbb8fa)3f4u$#1XU@_V@7fUPTOm)Q1t=xs2exWhn~kB9@x@L$^!?OJ^dc z%K3gwb3kFkWy8|(U}?lsv2LjkTk4m+E8P)GJK(ZXw`f>;X6bP7x#jF-ZK&yq(38WV zCq5VY+^LXhB&-~n@4KTlej07PuB{JC#PQpOnO%DnZ|_kZP^lqyz&M-UO*l+qQvXk= zRI2J962XK%=g|hSr>OM4PM?GJQYr_ikO`9{j|KE zUrN6ii0Jd@du|ymkf`svTocR)TgxKGa;(dyjD@G5s^aw0$l}Rhaj*v}DLR%PTRyik zvOE#e?+z(C!YmT zGQjFCoJXjS4UG&q-V4I`|9WWfDh!G=3=Z+FU~d{6n)u+*4hRNxp7G5yJ?M<=A1zO7lD;eo9Vrinh;FhjV@N%iiX|VZg6YwEUaKI-?aKt&k zQs{gdA7jWKdWS3-i~UQ!%f}*?GVtTlGZ)VXZI@qIE(vGVM$+pR)VIv(7&4Dv-V-r9 za6HgzTRa_siTI?pS$j<=pDX%0$TA z9ny5O5dx}0K&DKv)R2eK?DggGWTs|K3uUJ=h ztd_3ShV=VF%6($jp}2{Mpa(to&cH}LcvLo(|jR z@@onSSN#HX-|?z38J5kofe$YuuLtHso^#l!#IDz)ha899vrt2)2WAAl7sJ>na|_fA zq!xr}!cdFlngtLc2C_f}f#ha6!lWKUGKj%I4nvEaPzT?g)JmlYJB6!xV8= zOUlyrX>F0D2DZ~~_AMk#&m7ZUNJ@^V?Os*QDR9guK_?^`-{ZtR+44wfBoH2@kIKxR zfReG&adhdERTl+tkcov%gH6qBYS=W)2H_Z9%WPWMbj+sv4eO-`y=9R~lY9$3DTi+$ zC?ky>h;3j?V7ntXJ`$~D&p7cRwmjSCBDIM0X8LHM8BNL!halz=Gj~g3_!LejkGqT) ze%yciSiIqyQA!&d7sm18kH?gH3rugqaM(j4K~g+w0Ur%w_KcTFw&WY>LJVl@6G&_{ zY*WkBxoYCgsdE)aYb?B#1ao)@-|$(tj2orKb)7R9lk zO}>2ICSRKm&`f;ze25$s&>5UjtfFfEW^v~9Xjwb&oXbJ$?%mR~_{6ck3i#~b)+Yz+M7QL#eJBa~&E=$w z^y=-yf{#^`dOR&9Mr|Nxt1r^%k@8Ny=W+wNd>z}=Rq*w5xgHC?8{*%M_|^n+1ts4U z$UTRB4jwKkxiklIHy?Nl(XN$tC&bm2^Z?s7|R%+hg1cSdai)YhgxsF~cy8m*e%wtOiO?zgNG_?I zO7O)~MMpjfBTe-G;m_3n#dF08jN%A?6j;Rou!>{bjtJdcNlHJR_;@V>B~q(=QXM}D zTbL^il=6?p1xyTc#r$J)W&GoF<@i+uGKJDW{y3>~+45`}!3iPp(h_zJl$LE}Z0znmk$;#{u zIQhZ2ZCfZQT*plP+?IyHr|@T(#nq_s{*;mV1nq!q8e<9CC7_i`I6m_6)c7a)rzAXW zTmPr|XC(i*8h$8HBd$Sn)t<8{`9Hh;YKGB&f<`}oG8JY5ru-IzDYZVgRHE(U2Wqh= zntSbY-%^U+~Ze{F($+e6I@{Mk*R8U))za=XC$QbT$&QKyWDhMi3f z`;bun6o`>f8uID!ptSS(KT3%2V0`3nH{VmzPN?LN+6AyGN!JGWxoul9UkbeidVeyl zpa04D(I@++U{B3~=7(TU#9ypTDgO%y`57^v%{B5I*abOUf8r*m}zhZ+LNMpbgxLi*)D2 zR{4@e-KDsM$u>Uu2?gq{Qsz)zlZRu2NvLBdq2g4mkk6U3Fr~HD7p#lgOFx0yTG)~* zY(AEc_EP7z`Q&WMr+jNZDS15msd;EFZTtp#RDMDpxbbXrJ$7=7noFeEC$(+r=5R2Q z=O(J=@>1Gp*Vc3cdE>{W7EQjT5k(5@#oQLvm;SG<`EJQkNud|hE&jHM2((FH>idYk7OmB*a0WMcFcnv zzWT}OR==pU6xJ5M8tEl1kihZl#H~OI$c{M;3N+3?TtvD2c4#DFs!$zX$Z(9fp%|WY z0O7Dw5Mh$KScI$e`A1}YgN$#I@y9S`NQ(ENx>BZuF&w_b$x1N~shS;*(};y!;E)uxrc#^cWrl@5RUPm~;#S6B@k*U*D;(0kJC9#o99}Z&CAwbDh_!9)~ z^Xupuq`EE$9r*nEqf>s!t-Bi9>Kmb|g*NK2dFN#sNBp^ciQAi`4stMkt;is$1h#(K zbJ}0HP0)mzScvR4Qg74f+pN7OuIjbkaV*{<9gfPFtdi-LPN?`{$_Yu!PGw9<`b8e( zi;9Avd}wwXrLbwEZVDxgBez(Zs5FS(Nx%|(f+>fND+-#SeIVxCFCM}rJu`Hkm1zut z0Dp%AGCjXb#m<2tDVPy`C7}%zOn8lglu?k^E~O;EB$p-CG7#8f>Sl^l#|CA_^n2oP zfT)bdHHu?_4pwJ&u+w{XY(oZ(ER0PTrrD&o?rmZ=@ylxi1v6*&!KpoMd zI0C44@lhS0PVCsPqv}!22S*c~^z`B6Boy{KN$|6??3`H z8L?-Mpq)^4HBAR!J6oDNaD@eKE`#Qs&Zfo=Q9afO)C{$C;K~B%?(Gz4e+1g9W+oosrIo;VPEK%I=lM#a9S3TR(9duN=1Klx#i{9eX1)_vd zvF;nDrkO%TN06DMoT>lxlrVTm#Tw3%Lkhja z@gYq?zk(1a^hjYjsBIN0;JmvPCce_1RKJh59U+%2* zYnhlPl_Dia2$XizHL)D)li55ZM`T!s&O&~8`OUSENOgDmI+1P%+T5?`;5+?k&z{5g zN!HV@lYU)?>!hd#yVI|I)-}uk?MZ(boDrF-(d_NO*@Moe+IX<0`Zk2ax^~j9mMjL> z$$rBhUg&tqU9(_uC_aQt@4)7cDlV9*~%+Oj(vpQY%mRQ80`wWaWB z&$?I$8%8L~w#C;RznBnGAT)U|#ALhu%DPF{v|oYX07^b*E*wTW&c)OdW2Y`O?`ZYw z>ae^Ltc0VI&RHRt5RenEv3v{v?pUw1nYG#z+Q%uneU&_}7OuN?ZQ7~);U;IAAKb&{;UckKt3>?MpX(TmP7}x*;q`zQ>&b!fC z5cd@!L_NS9R#%}z$E1?S{t+4tX6h&;AfR;LBTLxLOJN;Vbn@EK+{pBvd7X&cLSTLw z(foEPQXCVC>mGx?FOlZ216Gbzvmt$TDZy$LDAlruBX8LuLdZ!j-dWy+hkr*i*D`MI* z=!iY_+!!xBN&zfv^&4i(zNkt$?jkBhV;2jV2-i;eX1F(?pBUbX*-j#JDM`c+AkpH_ z#NLI&RB)phf*m=U@1sieG04HE#*rMy#byL}-R7 zFmB8x_W+h!VWjb;)?SC zccQ~d&?NwoW)pGW_voq}M zjM(?CS@zN0R`nrk-EE_7-B@_jSSVeT)xTcW7A|XB<=3*iKXzSKGVzDqmqlsPVK~Qf zcc}gGwe;gS61I_M)|TA1n99QT~8OO98U`~ba--EwG^)9R2(W-b&)P)q0Eq&v#A z#`fEJmCFU8+?|W+TY06g?|EgiH8@>p%}LD?veL)Gxh*S$YndIZu7_p2`?jNO(Fj#}71eJ% zbM={TJPSpN<(&(vkf}I~Bb3RphV(P+7`r%wXkz28FrqIE?tEjkBto+vXrchSznx&7q)+`T%YId)}sdLS;mkvcbL~SRF zD$@Gc_xt;z`5l*gNyp=vo0%2S{JocZf1FbotXe4#m+ZWqzjv{hDe{1D=vRS zdsX`l{qlI&(Y~Ns@Psu5o152SU$QOvzIi<4+!ZPB2$^;-D3D-oKHz{f`be*4`jIum z(P&|LsH#0&NH-W+(D7`F*%H%=W?A}?8^fU^CqtIsVxC!1OU6cqF4wg1$bCDPSN?ka zEA>%J?l(JVcKubyU+?>ieW*j83931vyfSayTpl(<&&*A8_414B4ZY!p-nE8)bYRpr zf^gP!9G58Dy}B>d0Ua&e5m^sSnv!1mWfM9ln$@wodwuuO@b07Qy9dL&2P3>FoS9tl_MrD0bR)|75;m%`GJCYP@Ev!=ij5s(0) zlro5tlc@Seras5Ca0qqGDP6ZWgzXJ0+K9bl-QE?pcSY>o>-Ixo`=N;aFwVc4?MwYP z&85+DXUI~rk-h3TGdTTp+Xw?z%+mmj(K$ddBakx)r-Oc>ED2C19ADcyWm;0~hUpK|oiWcEaaipkmy{J80)V^w4^+t-i*NYB^iw@uD z4n6i{r0A(dL)1~bd}4WEwK!C`=dvE))S4B|>bd^Z_fNciBJ}tZq2o`4dY=sSJQaFs zC}bn$n-~pg94;L$3>WQOG~BV22K(PQa`nhc?TwPq;8X8ep1$*0vt869MWb!;`LLm6 zq5lK4nK^6STzu1940mhJ74u~?PLv%D;!?%lkfU$S(tq2M9V*~MZeNI>3C*0tXuKfn zqd|FI)=#f~`5}7cKO(=2b8e5w7uC0mt8jd}cn3kN5(ZKUEH9f@%{L0A;fb}y6tPvV z+Zw{Qh85+?NW|8$ZaWaR9k||dLlv?o zsr9<9a9vlZ_P}*#$aX~PK5JnxFKnsZ96Ku>8*SYm+!d;QG~{?}&GI;9l@qp9rv^sJ zkgW_ogr$$=m|ccjHczb7hnl)VUH*{m`9xN%|Dsv-%l_qw)za0;)!`d`A=`1u-%)ni za2Lq|cZiq0dbwk{c4cI>ZnX#sasf}JF^7=I*ipVnFRkHa-9=rQ1?HyM@1`;tl6j=QnCr^Z-u;mkYz1BDU@)%axuND%`Ph z{yQ(c^}_njBjKG#B0C2{JC23&AC06x7BW1Bg~XT@HCO<&qR^FQD2S4>Y22P_4q1vL zh7urucHM#j$e*q7%3ho;t^G6qcV2w!#kGun==1_EV1K>nm7?{WhHy?pB&TUTXJX-<@Fms-T>1bT=L+T z)xhw=nN7b&iERbYe{5`L$dGu+VAhUx%g&pYouvNQh)Z8{xvbsc%#LK>JuL8!mFHGQ zLRlT_mfbfkyIJ5p4>tD8fXF_5DG4{7zYG%wQq4WY!1298jZi!p7=f-**78e8eeKtx>)0RkUeX@3t2z4t1oPBc~8@V>k$$7U%aHN;|ncd6#Z{qyPvD&-f3!>tjFbu zUFrDvUUTu}ZpFJ9S+Kub-R1CTxbN$_((rY|+9mhw;eL=&>}gQ^pjAiqy(W14t+7ir zRnPq}y?Clv@xvA!**m)o@cECr1Lf0M+}}0GrZbd(*VxsJk8nv(BfNf;(Pezj%&j{H zEAjP@>bB>K6#uxV3HE=glc8w;RIi$BSN~IshFo^5JSz3iII{mtqxIA(erC0JDiuE~ zrgHtPQbjJc3bHq-JiF9CYt^7|Kii|7DpdSjWtz%S{5*p~{yaxTE`=2G=f$e28uiaB zG$`%Q>$TGs#m{$|ruB*q6@}c;tKhO>p^zIHG8B6wOZ%Kbu~Ag=oLX^LMq%!%RdBg$ zpfGpM+UN2WckRrcCnI}4vlpqJbE@x_QOWPtYK2C{-L^)dM)6U(4EB#IRH*4kH7c?< zDq!~;a6ZkA3k}cLfh(}-RGJpF^cV5F_q5_)b1Ds|22_a6p5zYgxJGhIzNNiiY3~-A zlyJg9d?0C}$JsRnF0cm$zV(q}LV5{TX0eb;*i_7>f?OPq$ZvhTsV9}7b0hVnQi{^y zZQ~|Nn1-~|x2BA8ssn0X6F-Nbom1mnf|}RGzxDWrP7j<*Fa*@>TmoED)C;EtaN;|l zO3HI=Uw)|!I0qJQ-^AGhg?M=uT0{9s*D`taKH^le2a4e^H|hC;EP0*?C(jdlK_G*o z!6||wF;uE4wLa8E7fbHrC6X=qPLaqXN z0M-|ExTH&doWtc?U)U<;ks^aYj9T_UmSi{1#bo$uBp0bZOa|c-bAD05Td&ztWDqiw zV?C(~kk3lUAn0GvGZ_TDld>pG1|fT^41!-DpO8#CAn`$G)?3j~{+vOu1`9jkFSL`fiwxRu7COkb8-~A(_RQitwAg5g^X45@shh%1;TVOFA?6@9 zyNC}jvF=2+IAh`pkHNi9I1Uqx8z8tufUPO$u<7?@IA;aLE%*ti;w+(L^BAIOqIv^ufFq44M7cZ!yddp(p zm-j_&%@=#4#`MKAVBtk=J1+MA*qXIewOk&~ZNTS3FYpX=dDH;(3n*7^)L_4r)3Boc zj^QmsB&QP$5VH;p5a6uLumE*mH{nygn*PT21295RMR{h5?$DW@!Zck(Nt}ugyYL#I?PB&gWgpT;k+TgqC}9H)$KeGm ztMQ3LfIOy|p5h`VswwV&qP~v0-I-SuI#5znkF$d zjeP843be`!e@ey|$e1T%kmiU240wTd#l@{GHg}Wrw3IwZUQ;+q#t9fPZQrqh!Qn@{ z1`mq&0@Hq)xIM)t^t<#)kW*A>$2PwHj6T`U5@{6@RU47M4W8ZJ}J3u^k zHm^kG25b{;hEgE0Lvadf_J|iH?-l+Sf$`6IGthd#Pi3`<(uL&iFoOhWQiD&Ma^=!2N?X;}^%~6}NSUm#Z&Uzg&N@ekmt-G@@&sSL066 zkfkc3cg|~~oaRfW&zlx~L4Mw}#?@fLW#!Myzoh%TZed55OOF=SieDjpUYN`ONMn?B z$?iHhnPrW$BCuTbrR>jVFBC;&Id|n2S@AsuCkMj{wvTEnWPP%`=na|iCDTWmTv__v z4$heSm1FbT`)MXw{-X0fhu21lL1qbNY;brjXWXOLMxjntvUGfd!z+0F9=$foz+Swo ze?cZYD2wWh8wzH_t)?3)W+On6xuF*A8qS=tp%v{qE+daFbG|3W-FQUFW!a(FA)1kO z*Z8PRmKM#-*-((>HZ(7*$bz&GU(Ku<4y;2hvl2IaL(i;WJXr6s*c%h_WP17`N{>oM zmfNQE4Ha3C9uiSAtA-NMGOLafVNo^$a_$M)Q?k2?JXtY|Pw$+94fb8rD8Bb}iSL2O z?veY(o;2B>r6&NzWP5_%d-U2!*UGd@BO4rE%Y*mmwP8<_?U#wc_RGX@`(^jjRkGT} zUH3V>z(b99VopSs@#B)JW!pDu?{W%R@lrQvAhAnr8*4J dict: + return {"text": self.text, "start": round(self.start, 3), "end": round(self.end, 3)} + + +def normalize(token: str) -> str: + """Forme comparable d'un mot : minuscules, sans accents ni ponctuation.""" + decomposed = unicodedata.normalize("NFKD", token.lower()) + stripped = "".join(c for c in decomposed if not unicodedata.combining(c)) + return re.sub(r"[^a-z0-9]", "", stripped) + + +def display_tokens(text: str) -> list[str]: + """Mots à afficher ; la ponctuation isolée est rattachée au mot précédent.""" + tokens: list[str] = [] + for raw in text.split(): + if tokens and not normalize(raw): + tokens[-1] += raw if raw in ",.!?;:…" else f" {raw}" + else: + tokens.append(raw) + return tokens + + +def align_words(display_text: str, spoken: list[Word], total_duration: float | None = None) -> list[Word]: + """Attribue à chaque mot affiché un début et une fin tirés des mots prononcés.""" + tokens = display_tokens(display_text) + if not tokens: + return [] + if not spoken: + return _spread(tokens, 0.0, total_duration or len(tokens) * 0.4) + + timed: list[Word | None] = [None] * len(tokens) + matcher = difflib.SequenceMatcher( + a=[normalize(t) for t in tokens], b=[normalize(w.text) for w in spoken], autojunk=False + ) + for block in matcher.get_matching_blocks(): + for k in range(block.size): + src = spoken[block.b + k] + timed[block.a + k] = Word(tokens[block.a + k], src.start, src.end) + + # Mots non appariés : répartis dans le trou entre leurs voisins datés + end_of_speech = max(spoken[-1].end, total_duration or 0.0) + i = 0 + while i < len(tokens): + if timed[i] is not None: + i += 1 + continue + j = i + while j < len(tokens) and timed[j] is None: + j += 1 + gap_start = timed[i - 1].end if i > 0 else 0.0 + gap_end = timed[j].start if j < len(tokens) else end_of_speech + if gap_end - gap_start < 0.05 * (j - i): + gap_end = gap_start + 0.25 * (j - i) + timed[i:j] = _spread(tokens[i:j], gap_start, gap_end) + i = j + + words = [w for w in timed if w is not None] + # Monotonie stricte : un mot ne commence jamais avant la fin du précédent + for prev, cur in zip(words, words[1:]): + cur.start = max(cur.start, prev.start + 0.01) + cur.end = max(cur.end, cur.start + 0.05) + return words + + +def _spread(tokens: list[str], start: float, end: float) -> list[Word]: + """Répartit des mots sur un intervalle au prorata de leur longueur.""" + weights = [max(1, len(normalize(t))) for t in tokens] + total = sum(weights) + words, cursor = [], start + for token, weight in zip(tokens, weights): + duration = (end - start) * weight / total + words.append(Word(token, cursor, cursor + duration)) + cursor += duration + return words + + +@lru_cache(maxsize=1) +def _model(): + from faster_whisper import WhisperModel + + log.info("Chargement du modèle Whisper %s", config.WHISPER_MODEL) + return WhisperModel(config.WHISPER_MODEL, device="cpu", compute_type=config.WHISPER_COMPUTE_TYPE) + + +def _transcribe_sync(audio: Path, hint: str | None) -> list[Word]: + segments, _ = _model().transcribe( + str(audio), + language="fr", + word_timestamps=True, + initial_prompt=hint or None, + vad_filter=False, + beam_size=5, + ) + return [ + Word(w.word.strip(), float(w.start), float(w.end)) + for segment in segments + for w in (segment.words or []) + if w.word.strip() + ] + + +async def transcribe(audio: Path, hint: str | None = None) -> list[Word]: + """Mots prononcés et leurs instants (Whisper, exécuté hors de la boucle async).""" + return await asyncio.to_thread(_transcribe_sync, audio, hint) diff --git a/ffmpeg-service/app/api.py b/ffmpeg-service/app/api.py new file mode 100644 index 0000000..b629c0d --- /dev/null +++ b/ffmpeg-service/app/api.py @@ -0,0 +1,304 @@ +"""API HTTP du service : voix, rendu des Reels et récupération des fichiers.""" + +import asyncio +import base64 +import contextlib +import logging +import time +from pathlib import Path + +import httpx +from fastapi import Depends, FastAPI, Header, HTTPException +from fastapi.responses import FileResponse +from pydantic import BaseModel + +from . import align, config, jobs, proc, render, subtitles, tts, voices +from .audio import encode_preview +from .text import clean_text + +logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s: %(message)s") +log = logging.getLogger("reels") + +if not config.API_KEY: + raise RuntimeError("API_KEY doit être défini : le service refuse de démarrer sans clé.") + +# Un seul encodage à la fois : ils saturent le CPU, les suivants attendent +render_slot = asyncio.Semaphore(1) + + +@contextlib.asynccontextmanager +async def lifespan(_app: FastAPI): + removed = jobs.purge_expired() + log.info("Service prêt (%d dossier(s) expiré(s) purgé(s))", removed) + + async def purge_loop(): + while True: + await asyncio.sleep(600) + jobs.purge_expired() + + task = asyncio.create_task(purge_loop()) + yield + task.cancel() + + +app = FastAPI(title="SocialFlow Reels", lifespan=lifespan) + + +def require_key(x_api_key: str | None = Header(None)) -> None: + if x_api_key != config.API_KEY: + raise HTTPException(status_code=401, detail="Invalid API Key") + + +class TtsRequest(BaseModel): + text: str + tts_voice: str | None = None + tts_engine: str | None = "gemini" + tts_style: str | None = None + gemini_api_key: str | None = None + + +class ReelRequest(BaseModel): + video_base64: str | None = None + video_url: str | None = None + text: str | None = None + music_url: str | None = None + watermark_url: str | None = None + store_name: str | None = None + font_size: int = 64 + music_volume: float = 0.25 + tts_enabled: bool = False + tts_voice: str | None = None + tts_engine: str | None = "gemini" + tts_style: str | None = None + gemini_api_key: str | None = None + draw_text: bool = True + stabilize: bool = False + enable_ending_effect: bool = True + # Champs d'anciennes versions, acceptés et ignorés + music_id: str | None = None + word_duration: float | None = None + + +@app.get("/health", dependencies=[Depends(require_key)]) +async def health(): + return {"status": "ok", "version": 2} + + +@app.get("/voices", dependencies=[Depends(require_key)]) +async def list_voices(): + return voices.catalog() + + +@app.post("/preview-tts", dependencies=[Depends(require_key)]) +async def preview_tts(request: TtsRequest): + text = clean_text(request.text) + if not text: + raise HTTPException(status_code=400, detail="Texte vide après nettoyage (emojis et hashtags retirés)") + + job_id, workdir = jobs.new_job() + try: + track = await _synthesize(text, request.text, request, workdir) + preview = workdir / "preview.mp3" + await encode_preview(track.path, preview) + return { + "success": True, + "audio_base64": base64.b64encode(preview.read_bytes()).decode(), + "duration": track.duration, + "words": [w.to_dict() for w in track.words], + "engine": track.engine, + "voice": track.voice, + "warnings": track.warnings, + } + finally: + jobs.remove_job(job_id) + + +@app.post("/process-reel", dependencies=[Depends(require_key)]) +async def process_reel(request: ReelRequest): + """Produit le Reel ; le MP4 se récupère ensuite via GET /files/{job_id}/output.mp4.""" + job_id, workdir = jobs.new_job() + stats: dict[str, float] = {} + started = time.monotonic() + try: + async with render_slot: + return await _process(request, job_id, workdir, stats, started) + except HTTPException: + jobs.remove_job(job_id) + raise + except proc.CommandError as error: + jobs.remove_job(job_id) + log.error("Rendu %s en échec : %s", job_id, error) + raise HTTPException(status_code=500, detail=str(error)) from error + except Exception as error: + jobs.remove_job(job_id) + log.exception("Rendu %s en échec", job_id) + raise HTTPException(status_code=500, detail=f"Erreur de rendu : {error}") from error + + +@app.get("/files/{job_id}/{name}", dependencies=[Depends(require_key)]) +async def get_file(job_id: str, name: str): + path = jobs.job_file(job_id, name) + if not path: + raise HTTPException(status_code=404, detail="Fichier introuvable ou expiré") + return FileResponse(path, media_type="video/mp4" if name.endswith(".mp4") else None) + + +@app.delete("/jobs/{job_id}", dependencies=[Depends(require_key)]) +async def delete_job(job_id: str): + jobs.remove_job(job_id) + return {"success": True} + + +# --- Orchestration -------------------------------------------------------- + + +async def _synthesize(text: str, display_source: str | None, request, workdir: Path) -> tts.VoiceTrack: + try: + return await tts.synthesize( + text=text, + display_text=clean_text(display_source) or text, + engine=request.tts_engine, + voice=request.tts_voice, + style=request.tts_style, + gemini_api_key=request.gemini_api_key, + workdir=workdir, + ) + except Exception as error: + log.exception("Voix impossible à générer") + raise HTTPException(status_code=502, detail=f"La voix n'a pas pu être générée : {error}") from error + + +async def _download(url: str, target: Path, what: str, required: bool) -> bool: + try: + async with httpx.AsyncClient(timeout=httpx.Timeout(120, connect=15), follow_redirects=True) as client: + async with client.stream("GET", url, headers={"User-Agent": "Mozilla/5.0"}) as response: + response.raise_for_status() + with target.open("wb") as out: + async for chunk in response.aiter_bytes(1 << 20): + out.write(chunk) + return True + except Exception as error: + if required: + detail = f"Téléchargement de {what} impossible : {error}" + raise HTTPException(status_code=400, detail=detail) from error + log.warning("Téléchargement de %s impossible, on continue sans : %s", what, error) + return False + + +async def _process(request: ReelRequest, job_id: str, workdir: Path, stats: dict, started: float) -> dict: + step = time.monotonic() + video = workdir / "input.mp4" + if request.video_base64: + video.write_bytes(base64.b64decode(request.video_base64)) + elif request.video_url: + await _download(request.video_url, video, "la vidéo", required=True) + else: + raise HTTPException(status_code=400, detail="Aucune vidéo fournie") + + music = workdir / "music.audio" + has_music = bool(request.music_url) and await _download( + request.music_url, music, "la musique", required=False + ) + watermark = workdir / "watermark.png" + has_watermark = bool(request.watermark_url) and await _download( + request.watermark_url, watermark, "le logo", required=False + ) + info = await proc.probe(video) + if info.duration <= 0: + raise HTTPException(status_code=400, detail="Vidéo illisible (durée nulle)") + stats["download"] = time.monotonic() - step + + # --- Voix --- + step = time.monotonic() + track = None + spoken_text = clean_text(request.text) + if request.tts_enabled and spoken_text: + track = await _synthesize(spoken_text, request.text, request, workdir) + stats["tts"] = time.monotonic() - step + + plan = render.RenderPlan( + video=video, + video_duration=info.duration, + output=workdir / "output.mp4", + is_hdr=info.is_hdr, + music=music if has_music else None, + music_volume=request.music_volume, + voice=track.path if track else None, + voice_duration=track.duration if track else 0.0, + watermark=watermark if has_watermark else None, + ending_effect=request.enable_ending_effect, + keep_original_audio=info.has_audio, + ) + + # --- Sous-titres --- + step = time.monotonic() + font_size = max(48, round(request.font_size * 1.4)) + display = clean_text(request.text) + if request.draw_text and display: + captions = workdir / "captions.ass" + if track: + subtitles.write_captions(track.words, captions, offset=plan.voice_delay, font_size=font_size) + else: + words = await _caption_words_without_voice(display, video, info, plan) + subtitles.write_captions(words, captions, offset=0.0, font_size=font_size) + plan.captions = captions + if request.store_name and request.enable_ending_effect and has_watermark: + outro = workdir / "outro.ass" + subtitles.write_outro(request.store_name, outro, plan.logo_start, plan.total_duration) + plan.outro = outro + stats["subtitles"] = time.monotonic() - step + + # --- Stabilisation (1re passe) --- + step = time.monotonic() + if request.stabilize: + transforms = workdir / "transforms.trf" + try: + await proc.run(render.stabilize_detect_command(video, transforms), timeout=600) + plan.stabilize_transforms = transforms + except proc.CommandError as error: + log.warning("Stabilisation ignorée : %s", error) + stats["stabilize"] = time.monotonic() - step + + # --- Encodage --- + step = time.monotonic() + await proc.run(render.build_command(plan), timeout=1200) + stats["encode"] = time.monotonic() - step + stats["total"] = time.monotonic() - started + + duration = (await proc.probe(plan.output)).duration + log.info( + "Rendu %s terminé : %.1f s de vidéo, étapes %s", + job_id, + duration, + {k: round(v, 1) for k, v in stats.items()}, + ) + + # Seul le résultat est conservé jusqu'au téléchargement + for entry in workdir.iterdir(): + if entry != plan.output: + entry.unlink(missing_ok=True) + + return { + "success": True, + "job_id": job_id, + "output_path": f"/files/{job_id}/output.mp4", + "duration": duration, + "tts_engine": track.engine if track else None, + "tts_voice": track.voice if track else None, + "warnings": track.warnings if track else [], + "processing_stats": stats, + } + + +async def _caption_words_without_voice(display: str, video: Path, info, plan: render.RenderPlan): + """Sans voix de synthèse : calage sur la parole de la vidéo si elle en contient, + sinon texte réparti sur la durée (hors effet de fin).""" + if info.has_audio: + try: + spoken = await align.transcribe(video, hint=display) + if len(spoken) >= max(2, len(display.split()) // 3): + return align.align_words(display, spoken, info.duration) + except Exception as error: # noqa: BLE001 — on retombe sur la répartition + log.warning("Transcription de la vidéo impossible : %s", error) + end = plan.logo_start if plan.ending_effect and plan.watermark else plan.total_duration + return align.align_words(display, [], max(1.0, end - 0.5)) diff --git a/ffmpeg-service/app/audio.py b/ffmpeg-service/app/audio.py new file mode 100644 index 0000000..dbdb48c --- /dev/null +++ b/ffmpeg-service/app/audio.py @@ -0,0 +1,60 @@ +"""Traitement de la voix : filtrage, compression et niveau sonore constant.""" + +from pathlib import Path + +from . import proc + +# Coupe les basses inutiles, lisse la dynamique et amène la voix à -16 LUFS : +# elle reste intelligible par-dessus la musique sans saturer. +VOICE_CHAIN = ( + "highpass=f=80," + "acompressor=threshold=0.1:ratio=3:attack=5:release=120:makeup=2," + "loudnorm=I=-16:TP=-1.5:LRA=7," + "aresample=48000" +) + + +async def process_voice(source: Path, target: Path) -> None: + """Produit un WAV 48 kHz mono prêt à mixer.""" + await proc.run( + [ + "ffmpeg", + "-y", + "-hide_banner", + "-loglevel", + "error", + "-i", + str(source), + "-af", + VOICE_CHAIN, + "-ac", + "1", + "-ar", + "48000", + "-c:a", + "pcm_s16le", + str(target), + ], + timeout=120, + ) + + +async def encode_preview(source: Path, target: Path) -> None: + """MP3 de bonne qualité pour l'écoute dans le navigateur.""" + await proc.run( + [ + "ffmpeg", + "-y", + "-hide_banner", + "-loglevel", + "error", + "-i", + str(source), + "-c:a", + "libmp3lame", + "-b:a", + "160k", + str(target), + ], + timeout=120, + ) diff --git a/ffmpeg-service/app/config.py b/ffmpeg-service/app/config.py new file mode 100644 index 0000000..9add083 --- /dev/null +++ b/ffmpeg-service/app/config.py @@ -0,0 +1,32 @@ +"""Configuration lue dans l'environnement.""" + +import os +from pathlib import Path + +# Pas de clé par défaut : un service exposé avec une clé connue de tous +# n'est pas protégé. Le démarrage échoue si elle manque. +API_KEY = os.environ.get("API_KEY", "") + +TEMP_DIR = Path(os.environ.get("TEMP_DIR", "/tmp/ffmpeg_processing")) + +# Durée de conservation des fichiers produits (téléchargés par l'application +# puis supprimés ; la purge ne rattrape que les oublis). +FILE_TTL_SECONDS = int(os.environ.get("FILE_TTL_SECONDS", "3600")) + +GEMINI_TTS_MODEL = os.environ.get("GEMINI_TTS_MODEL", "gemini-2.5-flash-preview-tts") + +# Modèle Whisper utilisé pour caler les sous-titres sur la voix. +WHISPER_MODEL = os.environ.get("WHISPER_MODEL", "base") +WHISPER_COMPUTE_TYPE = os.environ.get("WHISPER_COMPUTE_TYPE", "int8") + +# Proxy sortant éventuel pour Edge TTS (aiohttp ne lit pas HTTPS_PROXY seul). +OUTBOUND_PROXY = os.environ.get("HTTPS_PROXY") or os.environ.get("https_proxy") or None + +# Rendu +WIDTH = 1080 +HEIGHT = 1920 +FPS = 30 +# La voix démarre après ce délai (le temps de capter l'attention). +VOICE_DELAY = 2.0 + +SUBTITLE_FONT = os.environ.get("SUBTITLE_FONT", "Montserrat") diff --git a/ffmpeg-service/app/jobs.py b/ffmpeg-service/app/jobs.py new file mode 100644 index 0000000..b5dbdce --- /dev/null +++ b/ffmpeg-service/app/jobs.py @@ -0,0 +1,44 @@ +"""Dossiers de travail des rendus et purge des fichiers oubliés.""" + +import re +import shutil +import time +import uuid +from pathlib import Path + +from . import config + +_SAFE_ID = re.compile(r"^[0-9a-f-]{36}$") +_SAFE_NAME = re.compile(r"^[\w.-]+$") + + +def new_job() -> tuple[str, Path]: + job_id = str(uuid.uuid4()) + workdir = config.TEMP_DIR / job_id + workdir.mkdir(parents=True) + return job_id, workdir + + +def job_file(job_id: str, name: str) -> Path | None: + """Fichier d'un job, ou None si l'identifiant ou le nom sont suspects.""" + if not _SAFE_ID.match(job_id) or not _SAFE_NAME.match(name): + return None + path = config.TEMP_DIR / job_id / name + return path if path.is_file() else None + + +def remove_job(job_id: str) -> None: + if _SAFE_ID.match(job_id): + shutil.rmtree(config.TEMP_DIR / job_id, ignore_errors=True) + + +def purge_expired() -> int: + """Supprime les dossiers plus vieux que FILE_TTL_SECONDS.""" + config.TEMP_DIR.mkdir(parents=True, exist_ok=True) + limit = time.time() - config.FILE_TTL_SECONDS + removed = 0 + for entry in config.TEMP_DIR.iterdir(): + if entry.is_dir() and entry.stat().st_mtime < limit: + shutil.rmtree(entry, ignore_errors=True) + removed += 1 + return removed diff --git a/ffmpeg-service/app/proc.py b/ffmpeg-service/app/proc.py new file mode 100644 index 0000000..185b97c --- /dev/null +++ b/ffmpeg-service/app/proc.py @@ -0,0 +1,78 @@ +"""Exécution asynchrone de FFmpeg/ffprobe : le serveur reste disponible +pendant un encodage (l'ancien subprocess.run bloquait toute l'API).""" + +import asyncio +import json +import logging +from dataclasses import dataclass +from pathlib import Path + +log = logging.getLogger(__name__) + + +class CommandError(RuntimeError): + """Commande externe en échec, avec la fin de sa sortie d'erreur.""" + + def __init__(self, cmd: list[str], returncode: int, stderr: str): + tail = "\n".join(stderr.strip().splitlines()[-15:]) + super().__init__(f"{cmd[0]} a échoué (code {returncode}) :\n{tail}") + self.returncode = returncode + self.stderr = stderr + + +async def run(cmd: list[str], timeout: float = 900) -> str: + """Lance une commande et renvoie sa sortie standard.""" + log.debug("exec: %s", " ".join(cmd)) + process = await asyncio.create_subprocess_exec( + *cmd, stdout=asyncio.subprocess.PIPE, stderr=asyncio.subprocess.PIPE + ) + try: + stdout, stderr = await asyncio.wait_for(process.communicate(), timeout) + except TimeoutError as error: + process.kill() + await process.wait() + raise CommandError(cmd, -1, f"délai de {timeout:.0f} s dépassé") from error + if process.returncode != 0: + raise CommandError(cmd, process.returncode, stderr.decode(errors="replace")) + return stdout.decode(errors="replace") + + +@dataclass +class MediaInfo: + duration: float + has_audio: bool + width: int = 0 + height: int = 0 + color_transfer: str = "" + + @property + def is_hdr(self) -> bool: + # HLG (iPhone) ou PQ (HDR10) + return self.color_transfer in ("arib-std-b67", "smpte2084") + + +async def probe(path: Path) -> MediaInfo: + out = await run( + [ + "ffprobe", + "-v", + "error", + "-print_format", + "json", + "-show_format", + "-show_streams", + str(path), + ], + timeout=60, + ) + data = json.loads(out) + streams = data.get("streams", []) + video = next((s for s in streams if s.get("codec_type") == "video"), {}) + duration = float(data.get("format", {}).get("duration") or video.get("duration") or 0) + return MediaInfo( + duration=duration, + has_audio=any(s.get("codec_type") == "audio" for s in streams), + width=int(video.get("width") or 0), + height=int(video.get("height") or 0), + color_transfer=video.get("color_transfer") or "", + ) diff --git a/ffmpeg-service/app/render.py b/ffmpeg-service/app/render.py new file mode 100644 index 0000000..84c3a71 --- /dev/null +++ b/ffmpeg-service/app/render.py @@ -0,0 +1,217 @@ +"""Construction de la commande FFmpeg d'un Reel (fonction pure, testable).""" + +from dataclasses import dataclass +from pathlib import Path + +from . import config +from .subtitles import filter_path + +FADE_SECONDS = 2.0 +LOGO_SECONDS = 5.0 +# Silence laissé après la dernière phrase avant la fin de la vidéo +VOICE_TAIL = 0.8 +# Durée minimale de l'effet de fin (grand logo + nom du magasin) après la voix +OUTRO_MIN = 2.5 + +# HDR (HLG/PQ) → SDR BT.709 : sans cela, les vidéos iPhone sortent ternes +TONEMAP = ( + "zscale=t=linear:npl=100,format=gbrpf32le,zscale=p=bt709," + "tonemap=tonemap=hable:desat=0,zscale=t=bt709:m=bt709:r=tv,format=yuv420p" +) + + +@dataclass +class RenderPlan: + video: Path + video_duration: float + output: Path + is_hdr: bool = False + stabilize_transforms: Path | None = None + music: Path | None = None + music_volume: float = 0.25 + voice: Path | None = None + voice_duration: float = 0.0 + voice_delay: float = config.VOICE_DELAY + captions: Path | None = None + watermark: Path | None = None + outro: Path | None = None # nom du magasin (effet de fin) + ending_effect: bool = True + keep_original_audio: bool = False + + @property + def speech_end(self) -> float: + return self.voice_delay + self.voice_duration if self.voice else 0.0 + + @property + def total_duration(self) -> float: + """La vidéo s'allonge (dernière image figée) si la voix dure plus longtemps.""" + duration = self.video_duration + if self.voice: + duration = max(duration, self.speech_end + VOICE_TAIL) + if self.ending_effect and self.watermark: + duration = max(duration, self.speech_end + OUTRO_MIN) + return round(duration, 3) + + @property + def logo_start(self) -> float: + """Le grand logo n'arrive qu'une fois la voix terminée.""" + return max(0.0, self.total_duration - LOGO_SECONDS, self.speech_end) + + @property + def freeze_duration(self) -> float: + return max(0.0, self.total_duration - self.video_duration) + + +def build_command(plan: RenderPlan) -> list[str]: + total = plan.total_duration + logo_start = plan.logo_start + fade_start = max(0.0, total - FADE_SECONDS) + + cmd = ["ffmpeg", "-y", "-hide_banner", "-loglevel", "error", "-i", str(plan.video)] + index = 1 + music_idx = voice_idx = wm_idx = None + if plan.music: + # Musique bouclée : une piste plus courte que la vidéo ne coupe plus le son + cmd += ["-stream_loop", "-1", "-i", str(plan.music)] + music_idx, index = index, index + 1 + if plan.voice: + cmd += ["-i", str(plan.voice)] + voice_idx, index = index, index + 1 + if plan.watermark: + cmd += ["-i", str(plan.watermark)] + wm_idx, index = index, index + 1 + + graph: list[str] = [] + + # --- Vidéo --- + chain = [] + if plan.stabilize_transforms: + chain.append( + f"vidstabtransform=input={filter_path(plan.stabilize_transforms)}:smoothing=30:relative=1:zoom=5," + "unsharp=5:5:0.6:5:5:0.0" + ) + if plan.is_hdr: + chain.append(TONEMAP) + chain.append( + f"scale={config.WIDTH}:{config.HEIGHT}:force_original_aspect_ratio=increase:flags=lanczos," + f"crop={config.WIDTH}:{config.HEIGHT},setsar=1,fps={config.FPS}" + ) + if plan.freeze_duration > 0: + chain.append(f"tpad=stop_mode=clone:stop_duration={plan.freeze_duration:.3f}") + if plan.captions: + chain.append(f"subtitles='{filter_path(plan.captions)}'") + graph.append(f"[0:v]{','.join(chain)}[vbase]") + + current = "vbase" + if wm_idx is not None: + corner = "W-w-30:H-h-30" + if plan.outro and plan.ending_effect: + graph.append(f"[{wm_idx}:v]scale=200:-1,split=2[wm_small][wm_big0]") + graph.append("[wm_big0]scale=-1:300[wm_big]") + graph.append(f"[{current}][wm_small]overlay={corner}:enable='lt(t,{logo_start:.3f})'[vwm1]") + graph.append(f"[vwm1][wm_big]overlay=(W-w)/2:(H-h)/2-100:enable='gte(t,{logo_start:.3f})'[vwm2]") + current = "vwm2" + else: + graph.append(f"[{wm_idx}:v]scale=200:-1[wm_small]") + graph.append(f"[{current}][wm_small]overlay={corner}[vwm1]") + current = "vwm1" + if plan.outro and plan.ending_effect: + graph.append(f"[{current}]subtitles='{filter_path(plan.outro)}'[vout0]") + current = "vout0" + + tail = [] + if plan.ending_effect: + tail.append(f"fade=t=out:st={fade_start:.3f}:d={FADE_SECONDS}") + tail.append("format=yuv420p") + graph.append(f"[{current}]{','.join(tail)}[vout]") + + # --- Audio --- + mix = None + if voice_idx is not None: + delay_ms = int(plan.voice_delay * 1000) + graph.append(f"[{voice_idx}:a]aresample=48000,adelay={delay_ms}:all=1,apad[voice]") + if music_idx is not None: + graph.append(f"[{music_idx}:a]aresample=48000,volume={plan.music_volume:.3f}[music]") + + if voice_idx is not None and music_idx is not None: + # La musique baisse automatiquement quand la voix parle (ducking) + graph.append("[voice]asplit=2[vmix][vkey]") + graph.append("[music][vkey]sidechaincompress=threshold=0.02:ratio=8:attack=20:release=400[ducked]") + graph.append("[ducked][vmix]amix=inputs=2:duration=longest:normalize=0[mix]") + mix = "mix" + elif voice_idx is not None: + mix = "voice" + elif music_idx is not None: + mix = "music" + + audio_map: list[str] = [] + if mix: + # Niveau final recommandé par les réseaux sociaux (~ -14 LUFS) + audio_tail = ["loudnorm=I=-14:TP=-1.5:LRA=11", "aresample=48000"] + if plan.ending_effect: + audio_tail.append(f"afade=t=out:st={fade_start:.3f}:d={FADE_SECONDS}") + graph.append(f"[{mix}]{','.join(audio_tail)}[aout]") + audio_map = ["-map", "[aout]"] + elif plan.keep_original_audio: + audio_map = ["-map", "0:a:0"] + + cmd += ["-filter_complex", ";".join(graph), "-map", "[vout]", *audio_map] + cmd += [ + "-t", + f"{total:.3f}", + "-c:v", + "libx264", + "-preset", + "medium", + "-crf", + "19", + "-maxrate", + "10M", + "-bufsize", + "20M", + "-profile:v", + "high", + "-level", + "4.2", + "-g", + str(config.FPS * 2), + "-keyint_min", + str(config.FPS), + "-pix_fmt", + "yuv420p", + "-color_primaries", + "bt709", + "-color_trc", + "bt709", + "-colorspace", + "bt709", + "-c:a", + "aac", + "-b:a", + "192k", + "-ar", + "48000", + "-ac", + "2", + "-movflags", + "+faststart", + str(plan.output), + ] + return cmd + + +def stabilize_detect_command(video: Path, transforms: Path) -> list[str]: + return [ + "ffmpeg", + "-y", + "-hide_banner", + "-loglevel", + "error", + "-i", + str(video), + "-vf", + f"vidstabdetect=stepsize=32:shakiness=8:accuracy=15:result={filter_path(transforms)}", + "-f", + "null", + "-", + ] diff --git a/ffmpeg-service/app/subtitles.py b/ffmpeg-service/app/subtitles.py new file mode 100644 index 0000000..c5abe5f --- /dev/null +++ b/ffmpeg-service/app/subtitles.py @@ -0,0 +1,128 @@ +"""Sous-titres ASS : mot à mot, style « Reels » lisible sur tout fond.""" + +from pathlib import Path + +from . import config +from .align import Word + +# Couleurs ASS au format &HAABBGGRR +YELLOW = "&H0000E6FF" +WHITE = "&H00FFFFFF" +BLACK = "&H00000000" +SHADOW = "&H64000000" + +# Au-delà, une ligne déborde ou se lit mal en un coup d'œil +MAX_WORDS_PER_LINE = 3 +MAX_CHARS_PER_LINE = 18 +# Bas du texte à ~70 % de la hauteur : au-dessus de la légende et des boutons +# qu'Instagram et TikTok superposent en bas de l'écran. +MARGIN_V = 560 +POP_MS = 90 + + +def ass_time(seconds: float) -> str: + seconds = max(0.0, seconds) + centis = int(round(seconds * 100)) + hours, centis = divmod(centis, 360000) + minutes, centis = divmod(centis, 6000) + secs, centis = divmod(centis, 100) + return f"{hours}:{minutes:02d}:{secs:02d}.{centis:02d}" + + +def escape(text: str) -> str: + return text.replace("\\", "").replace("{", "(").replace("}", ")") + + +def header(styles: list[str]) -> str: + return ( + "[Script Info]\n" + "ScriptType: v4.00+\n" + f"PlayResX: {config.WIDTH}\n" + f"PlayResY: {config.HEIGHT}\n" + "ScaledBorderAndShadow: yes\n" + "WrapStyle: 2\n\n" + "[V4+ Styles]\n" + "Format: Name, Fontname, Fontsize, PrimaryColour, SecondaryColour, OutlineColour, " + "BackColour, Bold, Italic, Underline, StrikeOut, ScaleX, ScaleY, Spacing, Angle, " + "BorderStyle, Outline, Shadow, Alignment, MarginL, MarginR, MarginV, Encoding\n" + + "".join(f"{s}\n" for s in styles) + + "\n[Events]\n" + "Format: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text\n" + ) + + +def group_lines(words: list[Word]) -> list[list[Word]]: + """Découpe en lignes courtes, coupées de préférence à la ponctuation.""" + lines: list[list[Word]] = [] + current: list[Word] = [] + for word in words: + chars = sum(len(w.text) + 1 for w in current) + len(word.text) + if current and (len(current) >= MAX_WORDS_PER_LINE or chars > MAX_CHARS_PER_LINE): + lines.append(current) + current = [] + current.append(word) + if word.text.endswith((".", "!", "?", ",", ":", ";", "…")): + lines.append(current) + current = [] + if current: + lines.append(current) + return lines + + +def karaoke_line(line: list[Word], line_end: float) -> str: + """Chaque mot passe du blanc au jaune et « saute » légèrement quand il est dit.""" + parts = [] + line_start = line[0].start + for index, word in enumerate(line): + next_start = line[index + 1].start if index + 1 < len(line) else line_end + fill_cs = max(1, int(round((next_start - word.start) * 100))) + t0 = int(round((word.start - line_start) * 1000)) + pop = ( + f"\\fscx100\\fscy100" + f"\\t({t0},{t0 + POP_MS},\\fscx112\\fscy112)" + f"\\t({t0 + POP_MS},{t0 + 2 * POP_MS},\\fscx100\\fscy100)" + ) + parts.append(f"{{\\kf{fill_cs}{pop}}}{escape(word.text)}") + return " ".join(parts) + + +def write_captions( + words: list[Word], path: Path, *, offset: float, font_size: int, end_time: float | None = None +) -> None: + """Écrit le fichier ASS des sous-titres, décalés de `offset` secondes.""" + style = ( + f"Style: Caption,{config.SUBTITLE_FONT},{font_size},{YELLOW},{WHITE},{BLACK},{SHADOW}," + f"-1,0,0,0,100,100,0,0,1,6,3,2,80,80,{MARGIN_V},1" + ) + lines = group_lines(words) + events = [] + for index, line in enumerate(lines): + start = line[0].start + natural_end = line[-1].end + 0.25 + if index + 1 < len(lines): + end = min(max(natural_end, line[-1].end), lines[index + 1][0].start) + else: + end = max(natural_end, end_time - offset if end_time else natural_end) + end = max(end, start + 0.3) + events.append( + f"Dialogue: 0,{ass_time(start + offset)},{ass_time(end + offset)},Caption,,0,0,0,," + f"{karaoke_line(line, end)}" + ) + path.write_text(header([style]) + "\n".join(events) + "\n", encoding="utf-8") + + +def write_outro(store_name: str, path: Path, start: float, end: float) -> None: + """Nom du magasin sous le logo, en fondu, pendant les dernières secondes.""" + style = ( + f"Style: Outro,{config.SUBTITLE_FONT},72,{WHITE},{WHITE},{BLACK},{SHADOW}," + "-1,0,0,0,100,100,0,0,1,3,4,2,60,60,700,1" + ) + event = ( + f"Dialogue: 0,{ass_time(start)},{ass_time(end)},Outro,,0,0,0,,{{\\fad(1200,0)}}{escape(store_name)}" + ) + path.write_text(header([style]) + event + "\n", encoding="utf-8") + + +def filter_path(path: Path) -> str: + """Chemin utilisable dans un filtre FFmpeg (échappement de : et ').""" + return str(path).replace("\\", "/").replace(":", "\\:").replace("'", "\\'") diff --git a/ffmpeg-service/app/text.py b/ffmpeg-service/app/text.py new file mode 100644 index 0000000..f37e30f --- /dev/null +++ b/ffmpeg-service/app/text.py @@ -0,0 +1,20 @@ +"""Nettoyage du texte avant synthèse et affichage.""" + +import re + +import emoji + +_HIDDEN = str.maketrans("", "", "​") +_HASHTAG = re.compile(r"#[\wÀ-ɏ]+") +_URL = re.compile(r"https?://\S+") + + +def clean_text(text: str | None) -> str: + """Retire emojis, hashtags, liens et caractères invisibles.""" + if not text: + return "" + text = text.translate(_HIDDEN) + text = emoji.replace_emoji(text, replace="") + text = _URL.sub("", text) + text = _HASHTAG.sub("", text) + return " ".join(text.split()) diff --git a/ffmpeg-service/app/tts/__init__.py b/ffmpeg-service/app/tts/__init__.py new file mode 100644 index 0000000..0e8ce44 --- /dev/null +++ b/ffmpeg-service/app/tts/__init__.py @@ -0,0 +1,63 @@ +"""Synthèse vocale : moteur au choix, repli sur Edge, voix traitée et mots minutés.""" + +import logging +from dataclasses import dataclass, field +from pathlib import Path + +from .. import align, audio, proc +from ..align import Word +from . import edge, gemini + +log = logging.getLogger(__name__) + + +@dataclass +class VoiceTrack: + path: Path # WAV 48 kHz traité + duration: float + words: list[Word] # mots affichés, minutés depuis le début de la voix + engine: str + voice: str + warnings: list[str] = field(default_factory=list) + + +async def synthesize( + *, + text: str, + display_text: str, + engine: str | None, + voice: str | None, + style: str | None, + gemini_api_key: str | None, + workdir: Path, +) -> VoiceTrack: + warnings: list[str] = [] + raw: Path | None = None + spoken: list[Word] = [] + used_engine, used_voice = "edge", "" + + if (engine or "gemini") == "gemini": + if not gemini_api_key: + warnings.append("Clé Gemini absente : voix Edge utilisée à la place.") + else: + try: + raw, used_voice = await gemini.synthesize(text, voice, style, gemini_api_key, workdir) + used_engine = "gemini" + except Exception as error: # noqa: BLE001 — repli sur Edge + log.warning("Gemini TTS en échec, repli sur Edge : %s", error) + warnings.append(f"Gemini indisponible ({error}) : voix Edge utilisée à la place.") + + if raw is None: + raw, spoken, used_voice = await edge.synthesize(text, voice, style, workdir) + + processed = workdir / "voice.wav" + await audio.process_voice(raw, processed) + duration = (await proc.probe(processed)).duration + + if not spoken: + # Gemini ne donne pas le minutage : Whisper le retrouve dans l'audio + spoken = await align.transcribe(processed, hint=text) + + words = align.align_words(display_text, spoken, duration) + log.info("Voix prête : %s/%s, %.1f s, %d mots", used_engine, used_voice, duration, len(words)) + return VoiceTrack(processed, duration, words, used_engine, used_voice, warnings) diff --git a/ffmpeg-service/app/tts/edge.py b/ffmpeg-service/app/tts/edge.py new file mode 100644 index 0000000..e4defb4 --- /dev/null +++ b/ffmpeg-service/app/tts/edge.py @@ -0,0 +1,58 @@ +"""Synthèse vocale Edge (gratuite, fournit le minutage des mots).""" + +import logging +from pathlib import Path + +import edge_tts + +from .. import config +from ..align import Word +from ..voices import edge_fallbacks, resolve_edge_voice + +log = logging.getLogger(__name__) + +# Réglages de lecture par style (Edge n'interprète pas de consigne en texte) +STYLE_PROSODY = { + "dynamic": ("+8%", "+2Hz"), + "promo": ("+10%", "+3Hz"), + "calm": ("-8%", "-2Hz"), + "warm": ("-2%", "+0Hz"), +} + + +async def synthesize( + text: str, voice: str | None, style: str | None, workdir: Path +) -> tuple[Path, list[Word], str]: + """Génère la voix ; renvoie le MP3, les mots minutés et la voix utilisée.""" + rate, pitch = STYLE_PROSODY.get(style or "", ("+0%", "+0Hz")) + last_error: Exception | None = None + + for candidate in edge_fallbacks(resolve_edge_voice(voice)): + target = workdir / "edge.mp3" + try: + communicate = edge_tts.Communicate( + text, + candidate.id, + rate=rate, + pitch=pitch, + # edge-tts 7 ne renvoie plus que les phrases par défaut + boundary="WordBoundary", + proxy=config.OUTBOUND_PROXY, + ) + words: list[Word] = [] + with target.open("wb") as out: + async for chunk in communicate.stream(): + if chunk["type"] == "audio": + out.write(chunk["data"]) + elif chunk["type"] == "WordBoundary": + start = chunk["offset"] / 10_000_000 + words.append(Word(chunk["text"], start, start + chunk["duration"] / 10_000_000)) + if target.stat().st_size == 0: + raise RuntimeError("audio vide") + log.info("Edge TTS : voix=%s, %d mots minutés", candidate.id, len(words)) + return target, words, candidate.id + except Exception as error: # noqa: BLE001 — on essaie la voix suivante + log.warning("Edge TTS %s en échec : %s", candidate.id, error) + last_error = error + + raise RuntimeError(f"Aucune voix Edge disponible : {last_error}") diff --git a/ffmpeg-service/app/tts/gemini.py b/ffmpeg-service/app/tts/gemini.py new file mode 100644 index 0000000..a4c6dd2 --- /dev/null +++ b/ffmpeg-service/app/tts/gemini.py @@ -0,0 +1,102 @@ +"""Synthèse vocale Gemini (API generateContent, modalité audio).""" + +import base64 +import logging +import re +from pathlib import Path + +import httpx + +from .. import config, proc +from ..voices import resolve_gemini_voice, style_instruction + +log = logging.getLogger(__name__) + +API_ROOT = "https://generativelanguage.googleapis.com/v1beta/models" +FALLBACK_MODEL = "gemini-2.5-flash-preview-tts" + + +class GeminiError(RuntimeError): + pass + + +def build_prompt(text: str, style: str | None) -> str: + """Texte envoyé au modèle : la consigne de style précède le texte à lire.""" + instruction = style_instruction(style) + return f"{instruction} :\n{text}" if instruction else text + + +async def _request(model: str, prompt: str, voice: str, api_key: str) -> dict: + payload = { + "contents": [{"parts": [{"text": prompt}]}], + "generationConfig": { + "responseModalities": ["AUDIO"], + "speechConfig": {"voiceConfig": {"prebuiltVoiceConfig": {"voiceName": voice}}}, + }, + } + async with httpx.AsyncClient(timeout=120) as client: + # Clé en en-tête : dans l'URL, elle finissait dans les journaux + response = await client.post( + f"{API_ROOT}/{model}:generateContent", + json=payload, + headers={"x-goog-api-key": api_key}, + ) + if response.status_code != 200: + raise GeminiError(f"Gemini {model} : HTTP {response.status_code} {response.text[:300]}") + return response.json() + + +async def synthesize( + text: str, voice: str | None, style: str | None, api_key: str, workdir: Path +) -> tuple[Path, str]: + """Génère la voix ; renvoie un WAV brut et le nom de la voix utilisée.""" + gemini_voice = resolve_gemini_voice(voice).id + prompt = build_prompt(text, style) + model = config.GEMINI_TTS_MODEL + log.info("Gemini TTS : modèle=%s voix=%s style=%s (%d car.)", model, gemini_voice, style, len(text)) + + try: + data = await _request(model, prompt, gemini_voice, api_key) + except GeminiError as error: + if model == FALLBACK_MODEL or "HTTP 404" not in str(error): + raise + log.warning("Modèle %s indisponible, repli sur %s", model, FALLBACK_MODEL) + data = await _request(FALLBACK_MODEL, prompt, gemini_voice, api_key) + + try: + part = next(p for p in data["candidates"][0]["content"]["parts"] if "inlineData" in p)["inlineData"] + except (KeyError, IndexError, StopIteration) as error: + raise GeminiError(f"Réponse Gemini sans audio : {str(data)[:300]}") from error + + audio = base64.b64decode(part["data"]) + mime = part.get("mimeType", "") + raw = workdir / "gemini_raw.wav" + + if "wav" in mime: + raw.write_bytes(audio) + else: + # PCM brut 16 bits (« audio/L16;codec=pcm;rate=24000 ») + rate_match = re.search(r"rate=(\d+)", mime) + pcm = workdir / "gemini.pcm" + pcm.write_bytes(audio) + await proc.run( + [ + "ffmpeg", + "-y", + "-hide_banner", + "-loglevel", + "error", + "-f", + "s16le", + "-ar", + rate_match.group(1) if rate_match else "24000", + "-ac", + "1", + "-i", + str(pcm), + str(raw), + ], + timeout=60, + ) + pcm.unlink(missing_ok=True) + return raw, gemini_voice diff --git a/ffmpeg-service/app/voices.py b/ffmpeg-service/app/voices.py new file mode 100644 index 0000000..ab08bfb --- /dev/null +++ b/ffmpeg-service/app/voices.py @@ -0,0 +1,131 @@ +"""Catalogue des voix et des styles de lecture.""" + +from dataclasses import dataclass + + +@dataclass(frozen=True) +class Voice: + id: str + label: str + gender: str # "female" | "male" + + +# Voix natives de Gemini TTS (toutes multilingues, françaises comprises). +GEMINI_VOICES: tuple[Voice, ...] = ( + Voice("Kore", "Kore — ferme", "female"), + Voice("Aoede", "Aoede — légère", "female"), + Voice("Leda", "Leda — jeune", "female"), + Voice("Zephyr", "Zephyr — lumineuse", "female"), + Voice("Callirrhoe", "Callirrhoe — décontractée", "female"), + Voice("Autonoe", "Autonoe — lumineuse", "female"), + Voice("Despina", "Despina — douce", "female"), + Voice("Erinome", "Erinome — claire", "female"), + Voice("Laomedeia", "Laomedeia — enjouée", "female"), + Voice("Achernar", "Achernar — tendre", "female"), + Voice("Gacrux", "Gacrux — mûre", "female"), + Voice("Pulcherrima", "Pulcherrima — affirmée", "female"), + Voice("Vindemiatrix", "Vindemiatrix — délicate", "female"), + Voice("Sulafat", "Sulafat — chaleureuse", "female"), + Voice("Charon", "Charon — informative", "male"), + Voice("Puck", "Puck — enjouée", "male"), + Voice("Fenrir", "Fenrir — enthousiaste", "male"), + Voice("Orus", "Orus — ferme", "male"), + Voice("Enceladus", "Enceladus — soufflée", "male"), + Voice("Iapetus", "Iapetus — claire", "male"), + Voice("Umbriel", "Umbriel — décontractée", "male"), + Voice("Algieba", "Algieba — veloutée", "male"), + Voice("Algenib", "Algenib — rocailleuse", "male"), + Voice("Rasalgethi", "Rasalgethi — informative", "male"), + Voice("Alnilam", "Alnilam — ferme", "male"), + Voice("Schedar", "Schedar — posée", "male"), + Voice("Achird", "Achird — amicale", "male"), + Voice("Zubenelgenubi", "Zubenelgenubi — naturelle", "male"), + Voice("Sadachbia", "Sadachbia — vive", "male"), + Voice("Sadaltager", "Sadaltager — experte", "male"), +) + +EDGE_VOICES: tuple[Voice, ...] = ( + Voice("fr-FR-VivienneMultilingualNeural", "Vivienne", "female"), + Voice("fr-FR-DeniseNeural", "Denise", "female"), + Voice("fr-FR-EloiseNeural", "Eloise", "female"), + Voice("fr-FR-RemyMultilingualNeural", "Rémy", "male"), + Voice("fr-FR-HenriNeural", "Henri", "male"), +) + +# Consignes de lecture ajoutées au texte envoyé à Gemini. +STYLES: dict[str, tuple[str, str]] = { + "neutral": ("Neutre", ""), + "dynamic": ( + "Dynamique", + "Lis ce texte en français sur un ton dynamique et enthousiaste, avec un rythme " + "entraînant, comme une vidéo courte sur les réseaux sociaux", + ), + "warm": ( + "Chaleureux", + "Lis ce texte en français sur un ton chaleureux, souriant et proche, comme si tu conseillais un ami", + ), + "calm": ( + "Calme", + "Lis ce texte en français sur un ton calme, posé et rassurant, sans te presser", + ), + "promo": ( + "Promo", + "Lis ce texte en français comme une annonce promotionnelle énergique, en " + "insistant sur les offres et les prix", + ), +} + +_GEMINI_BY_ID = {v.id.lower(): v for v in GEMINI_VOICES} +_EDGE_BY_ID = {v.id: v for v in EDGE_VOICES} + + +def resolve_gemini_voice(voice: str | None) -> Voice: + """Voix Gemini à utiliser. + + Accepte aussi les anciens identifiants « fr-FR-Standard-X » encore stockés + dans des jobs (B/D = voix masculine). + """ + if voice and voice.lower() in _GEMINI_BY_ID: + return _GEMINI_BY_ID[voice.lower()] + if voice and voice.endswith(("-B", "-D")) or voice == "male": + return _GEMINI_BY_ID["charon"] + return _GEMINI_BY_ID["kore"] + + +def voice_gender(voice: str | None) -> str: + """Genre d'une voix, quel que soit le moteur d'origine.""" + if not voice: + return "female" + if voice.lower() in _GEMINI_BY_ID: + return _GEMINI_BY_ID[voice.lower()].gender + if voice in _EDGE_BY_ID: + return _EDGE_BY_ID[voice].gender + if voice == "male" or voice.endswith(("-B", "-D")): + return "male" + return "male" if any(name in voice for name in ("Remy", "Henri", "Paul")) else "female" + + +def resolve_edge_voice(voice: str | None) -> Voice: + """Voix Edge à utiliser ; une voix Gemini est remplacée par une voix Edge du même genre.""" + if voice in _EDGE_BY_ID: + return _EDGE_BY_ID[voice] + gender = voice_gender(voice) + return next(v for v in EDGE_VOICES if v.gender == gender) + + +def edge_fallbacks(primary: Voice) -> list[Voice]: + """Voix françaises de secours du même genre (jamais d'anglais).""" + same = [v for v in EDGE_VOICES if v.gender == primary.gender and v != primary] + return [primary, *same] + + +def style_instruction(style: str | None) -> str: + return STYLES.get(style or "neutral", STYLES["neutral"])[1] + + +def catalog() -> dict: + return { + "gemini": [v.__dict__ for v in GEMINI_VOICES], + "edge": [v.__dict__ for v in EDGE_VOICES], + "styles": [{"id": k, "label": v[0]} for k, v in STYLES.items()], + } diff --git a/ffmpeg-service/main.py b/ffmpeg-service/main.py index abe0963..3fdc2c3 100644 --- a/ffmpeg-service/main.py +++ b/ffmpeg-service/main.py @@ -1,1390 +1,10 @@ -from fastapi import FastAPI, HTTPException, Header -from pydantic import BaseModel -from typing import Optional -import uvicorn -import subprocess -import os -import uuid -import base64 -import requests -import shutil -from pathlib import Path -import edge_tts -import re -import emoji -import time +"""Point d'entrée uvicorn (`uvicorn main:app`).""" -app = FastAPI() - -API_KEY = os.environ.get("API_KEY", "default-key") -TEMP_DIR = Path("/tmp/ffmpeg_processing") -TEMP_DIR.mkdir(parents=True, exist_ok=True) - -# Set HOME for libass/fontconfig to ensure cache can be written -os.environ["HOME"] = "/tmp" -os.environ["XDG_CACHE_HOME"] = "/tmp/.cache" - - -# List available filters and fonts for debugging -def run_diagnostics(): - print("📋 Checking FFmpeg environment...") - try: - filters_out = subprocess.run( - ["ffmpeg", "-filters"], capture_output=True, text=True - ).stdout - has_subtitles = "subtitles" in filters_out - has_drawtext = "drawtext" in filters_out - print(f"✅ Filters found: subtitles={has_subtitles}, drawtext={has_drawtext}") - - print("📋 Available fonts (fc-list):") - subprocess.run(["fc-list"], check=True) - except Exception as e: - print(f"⚠️ Failed to check FFmpeg environment: {e}") - - -run_diagnostics() - - -@app.get("/debug-ffmpeg") -async def debug_ffmpeg(x_api_key: str = Header(None)): - if x_api_key != API_KEY: - raise HTTPException(status_code=401, detail="Invalid API Key") - try: - filters = subprocess.run( - ["ffmpeg", "-filters"], capture_output=True, text=True - ).stdout - fonts = subprocess.run(["fc-list"], capture_output=True, text=True).stdout - return { - "filters_summary": { - "subtitles": "subtitles" in filters, - "drawtext": "drawtext" in filters, - }, - "fonts": fonts.splitlines()[:50], # First 50 - "raw_filters_hint": filters[:500], - } - except Exception as e: - return {"error": str(e)} - - -# ... (existing imports) - - -def ensure_fonts(): - """Ensure Noto Color Emoji and other essential fonts are available.""" - print("🎨 Checking for Emoji fonts...") - - # Target directory for user fonts - font_dir = Path("/usr/share/fonts/truetype/noto") - if not font_dir.exists(): - try: - # Fallback to local user fonts if system dir is not writable - font_dir = Path("/tmp/.fonts") - font_dir.mkdir(parents=True, exist_ok=True) - except Exception: - font_dir = Path("/tmp/.fonts") - font_dir.mkdir(parents=True, exist_ok=True) - - emoji_font_path = font_dir / "NotoColorEmoji.ttf" - - if not emoji_font_path.exists(): - print("📥 Downloading Noto Color Emoji font...") - try: - url = "https://github.com/googlefonts/noto-emoji/raw/main/fonts/NotoColorEmoji.ttf" - response = requests.get(url, stream=True) - response.raise_for_status() - with open(emoji_font_path, "wb") as f: - shutil.copyfileobj(response.raw, f) - print(f"✅ Downloaded to {emoji_font_path}") - - # Update font cache - print("🔄 Updating font cache...") - subprocess.run( - ["fc-cache", "-f", "-v"], - stdout=subprocess.DEVNULL, - stderr=subprocess.DEVNULL, - ) - print("✅ Font cache updated") - except Exception as e: - print(f"⚠️ Failed to download emoji font: {e}") - else: - print(f"✅ Emoji font already present at {emoji_font_path}") - - -# Run font setup -ensure_fonts() - - -# Robust font detection -def get_font_path(): - # ... - possible_paths = [ - "/usr/share/fonts/truetype/dejavu/DejaVuSans.ttf", - "/usr/share/fonts/truetype/dejavu-core/DejaVuSans.ttf", - "/usr/share/fonts/TTF/DejaVuSans.ttf", - ] - for path in possible_paths: - if os.path.exists(path): - print(f"✅ Found font at: {path}") - return path - - # Fallback search - print("⚠️ Specific font not found, searching recursively in /usr/share/fonts...") - try: - found = [] - for root, dirs, files in os.walk("/usr/share/fonts"): - for file in files: - if file.endswith(".ttf"): - found.append(os.path.join(root, file)) - if found: - print(f"✅ Found {len(found)} fonts, using first: {found[0]}") - return found[0] - except Exception as e: - print(f"⚠️ Error searching for fonts: {e}") - - return "Sans" # Generic fallback - - -FONT_PATH = get_font_path() - - -class ReelRequest(BaseModel): - video_base64: Optional[str] = None - video_url: Optional[str] = None - text: Optional[str] = None - music_id: Optional[str] = None - music_url: Optional[str] = None - watermark_url: Optional[str] = None - store_name: Optional[str] = None - word_duration: float = 0.6 - font_size: int = 64 - music_volume: float = 0.25 - tts_enabled: bool = False - tts_voice: str = "fr-FR-VivienneMultilingualNeural" - tts_engine: str = "gemini" # "gemini" or "edge" - gemini_api_key: Optional[str] = None # Google Cloud API key for Gemini TTS - draw_text: bool = True - stabilize: bool = False # Stabilisation vidéo via vidstab - enable_ending_effect: bool = True - - -def clean_text_for_display(text: str) -> str: - """Removes emojis, hashtags, and hidden chars for display (text only).""" - if not text: - return "" - # 0. Remove BOM and other hidden characters - text = text.replace("\ufeff", "").replace("\u200b", "") - # 1. Remove emojis - text = emoji.replace_emoji(text, replace="") - # 2. Remove hashtags (e.g. #viral #fyp) - text = re.sub(r"#\w+", "", text) - # 3. Collapse multiple spaces - text = re.sub(r"\s+", " ", text).strip() - return text - - -def clean_text_for_tts(text: str) -> str: - if not text: - return "" - # 0. Remove BOM and other hidden characters - text = text.replace("\ufeff", "").replace("\u200b", "") - # 1. Remove emojis - text = emoji.replace_emoji(text, replace="") - # 2. Remove hashtags (e.g. #viral #reels) - text = re.sub(r"#\w+", "", text) - # 3. Cleanup whitespace - return " ".join(text.split()) - - -async def generate_tts_gemini( - text: str, - voice: str, - api_key: str, - audio_path: Path, - ass_path: Path, - display_text: Optional[str] = None, - delay: float = 0.0, -): - """Generate TTS audio using Gemini native TTS API (gemini-2.5-flash-preview-tts).""" - try: - import httpx - - print(f"\U0001f50a Gemini TTS request: voice={voice}, text_len={len(text)}") - - # Map Google Cloud TTS voice names (fr-FR-Standard-A/B/C/D) to Gemini native voices - # A/C = female, B/D = male - is_male = voice.endswith(("-B", "-D")) - gemini_voice = "Charon" if is_male else "Kore" - - print(f"\U0001f50a Mapped voice \'{voice}\' -> Gemini native voice \'{gemini_voice}\'") - - url = ( - "https://generativelanguage.googleapis.com/v1beta/models/" - f"gemini-2.5-flash-preview-tts:generateContent?key={api_key}" - ) - - payload = { - "contents": [{"parts": [{"text": text}]}], - "generationConfig": { - "response_modalities": ["AUDIO"], - "speech_config": { - "voice_config": { - "prebuilt_voice_config": { - "voice_name": gemini_voice - } - } - }, - }, - } - - async with httpx.AsyncClient(timeout=120.0) as client: - response = await client.post(url, json=payload) - response.raise_for_status() - data = response.json() - - # Extract audio from Gemini response - inline_data = data["candidates"][0]["content"]["parts"][0]["inlineData"] - mime_type = inline_data.get("mimeType", "audio/wav") - audio_bytes = base64.b64decode(inline_data["data"]) - - print(f"\U0001f50a Gemini audio received: {len(audio_bytes)} bytes, mime={mime_type}") - - # Gemini returns raw PCM (e.g. "audio/L16;codec=pcm;rate=24000") — parse rate from mime - sample_rate = 24000 - rate_match = re.search(r"rate=(\d+)", mime_type) - if rate_match: - sample_rate = int(rate_match.group(1)) - - pcm_path = audio_path.with_suffix(".pcm") - with open(pcm_path, "wb") as f: - f.write(audio_bytes) - - convert_cmd = [ - "ffmpeg", "-y", - "-f", "s16le", - "-ar", str(sample_rate), - "-ac", "1", - "-i", str(pcm_path), - "-codec:a", "libmp3lame", - "-qscale:a", "2", - str(audio_path), - ] - result = subprocess.run(convert_cmd, capture_output=True, text=True) - if result.returncode != 0: - print(f"❌ ffmpeg PCM->MP3 conversion failed (rc={result.returncode}):") - print(result.stderr[-1000:]) - raise RuntimeError(f"ffmpeg conversion failed: {result.stderr[-500:]}") - pcm_path.unlink(missing_ok=True) - - print(f"\u2705 Gemini TTS audio saved: {audio_path.stat().st_size} bytes") - - # Measure total audio duration - audio_duration = None - try: - duration_cmd = [ - "ffprobe", "-v", "error", - "-show_entries", "format=duration", - "-of", "default=noprint_wrappers=1:nokey=1", - str(audio_path), - ] - dur_proc = subprocess.run(duration_cmd, stdout=subprocess.PIPE, text=True) - audio_duration = float(dur_proc.stdout.strip()) - print(f"\u23f1\ufe0f TTS Audio Duration: {audio_duration:.2f}s") - except Exception as e: - print(f"\u26a0\ufe0f Could not measure TTS duration: {e}") - - # Gemini TTS does not return word boundaries — sync with ffsubsync - text_to_display = display_text if display_text else text - unsynced_srt_path = audio_path.with_suffix(".unsynced.srt") - synced_srt_path = audio_path.with_suffix(".synced.srt") - generate_unsynced_srt(text_to_display, unsynced_srt_path, total_duration=audio_duration) - run_ffsubsync(audio_path, unsynced_srt_path, synced_srt_path) - convert_srt_to_ass(synced_srt_path, ass_path, font_size=65, delay=delay) - print("\u2705 TTS synchronisation completed with ffsubsync") - return - - except Exception as e: - print(f"\u274c Gemini TTS failed: {e}") - raise - - -async def generate_tts_with_subs( - text: str, - voice: str, - audio_path: Path, - ass_path: Path, - display_text: Optional[str] = None, - delay: float = 0.0, -): - """Generate TTS audio with word-level synchronized subtitles. - - Uses edge_tts.Communicate.stream() to capture WordBoundary events, - providing millisecond-accurate subtitle timing. - """ - # Determine gender of requested voice to choose appropriate fallbacks - is_male = any(name in voice for name in ["Remy", "Henri", "Paul"]) - - if is_male: - fallback_voices = [ - voice, - "fr-FR-RemyMultilingualNeural", - "fr-FR-HenriNeural", - "fr-FR-PaulNeural", - ] - else: - fallback_voices = [ - voice, - "fr-FR-VivienneMultilingualNeural", - "fr-FR-VivienneNeural", - "fr-FR-DeniseNeural", - ] - - fallback_voices = list(dict.fromkeys(fallback_voices)) - - last_error = None - - for attempt_voice in fallback_voices: - try: - print(f"🔊 TTS attempt with voice: {attempt_voice}") - communicate = edge_tts.Communicate(text, attempt_voice) - - # Stream audio + word boundaries simultaneously - word_boundaries = [] - audio_chunks = [] - - async for chunk in communicate.stream(): - if chunk["type"] == "audio": - audio_chunks.append(chunk["data"]) - elif chunk["type"] == "WordBoundary": - # Offsets are in 100-nanosecond ticks, convert to seconds - offset_sec = chunk["offset"] / 10_000_000 - duration_sec = chunk["duration"] / 10_000_000 - word_boundaries.append( - { - "text": chunk["text"], - "offset": offset_sec, - "duration": duration_sec, - } - ) - - # Write audio to file - if audio_chunks: - with open(audio_path, "wb") as f: - for audio_data in audio_chunks: - f.write(audio_data) - - if audio_path.exists() and audio_path.stat().st_size > 0: - print(f"✅ TTS audio saved: {audio_path.stat().st_size} bytes") - print(f"📍 Captured {len(word_boundaries)} word boundaries") - - # Log a few boundaries for debugging - for wb in word_boundaries[:5]: - print( - f" → '{wb['text']}' at {wb['offset']:.2f}s (dur: {wb['duration']:.2f}s)" - ) - - # Measure total audio duration via ffprobe for safety - audio_duration = None - try: - duration_cmd = [ - "ffprobe", - "-v", - "error", - "-show_entries", - "format=duration", - "-of", - "default=noprint_wrappers=1:nokey=1", - str(audio_path), - ] - dur_proc = subprocess.run( - duration_cmd, stdout=subprocess.PIPE, text=True - ) - audio_duration = float(dur_proc.stdout.strip()) - print(f"⏱️ TTS Audio Duration: {audio_duration:.2f}s") - except Exception as e: - print(f"⚠️ Could not measure TTS duration: {e}") - - # Use display_text for subtitle content if provided - text_to_display = display_text if display_text else text - - if len(word_boundaries) == 0: - print("⚠️ No word boundaries captured, falling back to ffsubsync") - unsynced_srt_path = audio_path.with_suffix(".unsynced.srt") - synced_srt_path = audio_path.with_suffix(".synced.srt") - generate_unsynced_srt(text_to_display, unsynced_srt_path, total_duration=audio_duration) - run_ffsubsync(audio_path, unsynced_srt_path, synced_srt_path) - convert_srt_to_ass(synced_srt_path, ass_path, font_size=65, delay=delay) - print(f"✅ TTS synchronisation completed with ffsubsync fallback") - return - - print("🎯 Using precise word-boundary timing from TTS engine") - generate_ass_from_word_boundaries( - word_boundaries, - text_to_display, - ass_path, - font_size=65, - total_duration=audio_duration, - delay=delay, - ) - print(f"✅ TTS synchronisation completed with word-boundary timing") - return - else: - print(f"⚠️ Audio file empty or missing with voice: {attempt_voice}") - - except Exception as e: - print(f"⚠️ TTS failed with voice {attempt_voice}: {e}") - last_error = e - - raise Exception(f"All TTS voices failed. Last error: {last_error}") - - -def generate_ass_from_word_boundaries( - word_boundaries: list, - display_text: str, - ass_path: Path, - font_size: int = 65, - total_duration: float = None, - delay: float = 0.0, -): - """Generate ASS subtitles using precise word-level timing from TTS engine. - - Groups words into readable chunks (~3 words or at punctuation) and uses - the real start/end timestamps from the TTS engine for each chunk. - Includes karaoke fill tags for word-level visual highlighting. - """ - if not word_boundaries: - return - - # Group word boundaries into chunks of ~3 words, or split at punctuation - chunks = [] - current_words = [] - current_boundaries = [] - current_start = word_boundaries[0]["offset"] - - for i, wb in enumerate(word_boundaries): - current_words.append(wb["text"]) - current_boundaries.append(wb) - is_last = i == len(word_boundaries) - 1 - # Split at punctuation or every 3 words (tighter sync with voice) - ends_sentence = wb["text"].rstrip().endswith((".", "!", "?", ":", ",")) - at_limit = len(current_words) >= 3 - - if is_last or ends_sentence or at_limit: - # End time = this word's offset + its duration - chunk_end = wb["offset"] + wb["duration"] - chunks.append( - { - "words": current_words, - "boundaries": current_boundaries, - "start": current_start + delay, - "end": chunk_end + delay, - } - ) - current_words = [] - current_boundaries = [] - # Next chunk starts at the next word's offset - if not is_last: - current_start = word_boundaries[i + 1]["offset"] - - # Extend the last chunk to total_duration if available - # Why: prevents the last subtitle from vanishing before audio ends - if total_duration and chunks: - chunks[-1]["end"] = max(chunks[-1]["end"], total_duration) - - # Add 50ms overlap between consecutive chunks to prevent flickering - for i in range(len(chunks) - 1): - chunks[i]["end"] = max(chunks[i]["end"], chunks[i + 1]["start"] + 0.05) - - # ASS Header (same karaoke style as convert_srt_to_ass) - header = f"""[Script Info] -ScriptType: v4.00+ -PlayResX: 1080 -PlayResY: 1920 -ScaledBorderAndShadow: yes - -[V4+ Styles] -Format: Name, Fontname, Fontsize, PrimaryColour, SecondaryColour, OutlineColour, BackColour, Bold, Italic, Underline, StrikeOut, ScaleX, ScaleY, Spacing, Angle, BorderStyle, Outline, Shadow, Alignment, MarginL, MarginR, MarginV, Encoding -Style: Default,Sans,{font_size},&H0000FFFF,&H00FFFFFF,&H00000000,&H80000000,-1,0,0,0,100,100,0,0,1,4,2,5,50,50,0,1 - -[Events] -Format: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text -""" - - events = "" - for chunk in chunks: - start_ts = format_ass_time(chunk["start"]) - end_ts = format_ass_time(chunk["end"]) - - karaoke_parts = [] - for wb in chunk["boundaries"]: - # duration in centiseconds, minimum 10cs to avoid zero - w_dur_cs = max(10, int(wb["duration"] * 100)) - sanitized = wb["text"].replace("{", "(").replace("}", ")") - karaoke_parts.append(f"{{\\kf{w_dur_cs}}}{sanitized}") - - karaoke_text = " ".join(karaoke_parts) - events += f"Dialogue: 0,{start_ts},{end_ts},Default,,0,0,0,,{karaoke_text}\n" - - with open(ass_path, "w", encoding="utf-8") as f: - f.write(header + events) - - print( - f"📄 Generated synced ASS: {ass_path.stat().st_size} bytes, " - f"{len(chunks)} chunks from {len(word_boundaries)} words" - ) - - -def generate_simple_ass( - text: str, - ass_path: Path, - font_size: int = 65, - total_duration: float = None, - delay: float = 0.0, -): - """Generate TikTok-style ASS subtitle file with karaoke highlight effect. - - Each word fills from white to yellow as it is spoken, with thick outline - for readability on any background. - """ - # Split text into word lists (3 words max per chunk for TikTok readability) - all_words = text.split() - chunks = [] # Each chunk is a list of words - current_chunk = [] - - for word in all_words: - current_chunk.append(word) - if len(current_chunk) >= 3 or word.endswith((".", "!", "?", ":")): - chunks.append(current_chunk) - current_chunk = [] - - if current_chunk: - chunks.append(current_chunk) - - # ASS Header — TikTok Karaoke Style - # PrimaryColour = Yellow (highlighted/spoken) &H0000FFFF (ASS BGR: 00,FF,FF = RGB FF,FF,00) - # SecondaryColour = White (before highlight) &H00FFFFFF - # OutlineColour = Black &H00000000 - # BackColour = Semi-transparent black &H80000000 - # Bold=-1, Outline=3, Shadow=1, Alignment=5 (center middle) - header = f"""[Script Info] -ScriptType: v4.00+ -PlayResX: 1080 -PlayResY: 1920 -ScaledBorderAndShadow: yes - -[V4+ Styles] -Format: Name, Fontname, Fontsize, PrimaryColour, SecondaryColour, OutlineColour, BackColour, Bold, Italic, Underline, StrikeOut, ScaleX, ScaleY, Spacing, Angle, BorderStyle, Outline, Shadow, Alignment, MarginL, MarginR, MarginV, Encoding -Style: Default,Sans,{font_size},&H0000FFFF,&H00FFFFFF,&H00000000,&H80000000,-1,0,0,0,100,100,0,0,1,4,2,5,50,50,0,1 - -[Events] -Format: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text -""" - - events = "" - current_time = delay - - # Calculate total characters across all chunks for proportional timing - total_chars = sum(len(w) for chunk in chunks for w in chunk) - if total_chars == 0: - total_chars = 1 - - if total_duration: - time_per_char = total_duration / total_chars - else: - # Fallback: ~80ms per character - time_per_char = 0.08 - - for chunk_words in chunks: - # Calculate chunk duration from its characters - chunk_chars = sum(len(w) for w in chunk_words) - chunk_duration = chunk_chars * time_per_char - - start_time = format_ass_time(current_time) - end_time = format_ass_time(current_time + chunk_duration) - - # Build karaoke text with \kf tags per word - # \kf = smooth fill from SecondaryColour (white) to PrimaryColour (yellow) - karaoke_parts = [] - for word in chunk_words: - # Word duration in centiseconds, proportional to character length - word_dur_cs = int((len(word) / chunk_chars) * chunk_duration * 100) - word_dur_cs = max(word_dur_cs, 10) # Min 0.1s per word - sanitized = word.replace("{", "(").replace("}", ")") - karaoke_parts.append(f"{{\\kf{word_dur_cs}}}{sanitized}") - - karaoke_text = " ".join(karaoke_parts) - events += ( - f"Dialogue: 0,{start_time},{end_time},Default,,0,0,0,,{karaoke_text}\n" - ) - - current_time += chunk_duration - - with open(ass_path, "w", encoding="utf-8") as f: - f.write(header + events) - - print( - f"📄 Generated ASS file (karaoke): {ass_path.stat().st_size} bytes, {len(chunks)} chunks, {len(all_words)} words" - ) - - -def generate_outro_ass( - text: str, ass_path: Path, start_time: float, end_time: float, font_size: int = 70 -): - """Generate a simple ASS subtitle for the store name outro, fading in at the end.""" - header = f"""[Script Info] -ScriptType: v4.00+ -PlayResX: 1080 -PlayResY: 1920 -ScaledBorderAndShadow: yes - -[V4+ Styles] -Format: Name, Fontname, Fontsize, PrimaryColour, SecondaryColour, OutlineColour, BackColour, Bold, Italic, Underline, StrikeOut, ScaleX, ScaleY, Spacing, Angle, BorderStyle, Outline, Shadow, Alignment, MarginL, MarginR, MarginV, Encoding -Style: Default,Sans,{font_size},&H00FFFFFF,&H00FFFFFF,&H00000000,&H00000000,-1,0,0,0,100,100,0,0,1,0,4,2,0,0,0,1 - -[Events] -Format: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text -""" - events = "" - start_str = format_ass_time(start_time) - end_str = format_ass_time(end_time) - - # Alignment 2 is bottom center. MarginV = 700 pushes it up appropriately below the center logo. - # \fad(2000,0) fades in over 2000ms. - sanitized = text.replace("{", "(").replace("}", ")") - events += f"Dialogue: 0,{start_str},{end_str},Default,,0,0,700,,{{\\fad(2000,0)}}{sanitized}\n" - - with open(ass_path, "w", encoding="utf-8") as f: - f.write(header + events) - - -def format_ass_time(seconds: float) -> str: - """Format seconds as ASS timestamp (H:MM:SS.ss).""" - hours = int(seconds // 3600) - minutes = int((seconds % 3600) // 60) - secs = int(seconds % 60) - centis = int((seconds % 1) * 100) - return f"{hours}:{minutes:02d}:{secs:02d}.{centis:02d}" - - -def format_srt_time(seconds: float) -> str: - hours = int(seconds // 3600) - minutes = int((seconds % 3600) // 60) - secs = int(seconds % 60) - millis = int((seconds % 1) * 1000) - return f"{hours:02d}:{minutes:02d}:{secs:02d},{millis:03d}" - -def generate_unsynced_srt(text: str, srt_path: Path, total_duration: float = 30.0): - all_words = text.split() - chunks = [] - current_chunk = [] - - for word in all_words: - current_chunk.append(word) - if len(current_chunk) >= 3 or word.endswith((".", "!", "?", ":")): - chunks.append(" ".join(current_chunk).strip()) - current_chunk = [] - if current_chunk: - chunks.append(" ".join(current_chunk).strip()) - - total_chars = sum(len(c) for c in chunks) or 1 - current_time = 0.0 - - with open(srt_path, "w", encoding="utf-8") as f: - for i, chunk in enumerate(chunks): - chunk_dur = (len(chunk) / total_chars) * total_duration - start = current_time - end = start + chunk_dur - current_time = end - - f.write(f"{i+1}\n") - f.write(f"{format_srt_time(start)} --> {format_srt_time(end)}\n") - f.write(f"{chunk}\n\n") - -def run_ffsubsync(audio_path: Path, unsynced_srt: Path, synced_srt: Path): - print(f"🔄 Running ffsubsync on {audio_path.name}...") - cmd = [ - "ffsubsync", - str(audio_path), - "-i", str(unsynced_srt), - "-o", str(synced_srt) - ] - try: - proc = subprocess.run(cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True) - if proc.returncode == 0: - print("✅ ffsubsync success") - else: - print(f"⚠️ ffsubsync error: {proc.stderr[:200]}") - shutil.copy(unsynced_srt, synced_srt) - except Exception as e: - print(f"⚠️ ffsubsync exception: {e}") - shutil.copy(unsynced_srt, synced_srt) - -def parse_srt_time(s: str) -> float: - s = s.strip() - parts = s.split(",") - ms = int(parts[1]) if len(parts) > 1 else 0 - h, m, sec = parts[0].split(":") - return int(h)*3600 + int(m)*60 + int(sec) + ms/1000.0 - -def convert_srt_to_ass(srt_path: Path, ass_path: Path, font_size: int = 65, delay: float = 0.0): - with open(srt_path, "r", encoding="utf-8") as f: - content = f.read().strip() - - blocks = content.split("\n\n") - chunks = [] - for block in blocks: - lines = block.split("\n") - if len(lines) >= 3: - time_str = lines[1] - if " --> " in time_str: - start_str, end_str = time_str.split(" --> ") - - start = parse_srt_time(start_str) + delay - end = parse_srt_time(end_str) + delay - text = " ".join(lines[2:]) - chunks.append({"start": start, "end": end, "text": text}) - - header = f"""[Script Info] -ScriptType: v4.00+ -PlayResX: 1080 -PlayResY: 1920 -ScaledBorderAndShadow: yes - -[V4+ Styles] -Format: Name, Fontname, Fontsize, PrimaryColour, SecondaryColour, OutlineColour, BackColour, Bold, Italic, Underline, StrikeOut, ScaleX, ScaleY, Spacing, Angle, BorderStyle, Outline, Shadow, Alignment, MarginL, MarginR, MarginV, Encoding -Style: Default,Sans,{font_size},&H0000FFFF,&H00FFFFFF,&H00000000,&H80000000,-1,0,0,0,100,100,0,0,1,4,2,5,50,50,0,1 - -[Events] -Format: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text -""" - events = "" - for chunk in chunks: - start_ts = format_ass_time(chunk["start"]) - end_ts = format_ass_time(chunk["end"]) - sanitized = chunk["text"].replace("{", "(").replace("}", ")") - - words = sanitized.split() - chunk_dur = chunk["end"] - chunk["start"] - karaoke_parts = [] - total_chars = sum(len(x) for x in words) or 1 - for w in words: - w_dur_cs = int((len(w)/total_chars) * chunk_dur * 100) - w_dur_cs = max(10, w_dur_cs) - karaoke_parts.append(f"{{\\kf{w_dur_cs}}}{w}") - - karaoke_text = " ".join(karaoke_parts) - events += f"Dialogue: 0,{start_ts},{end_ts},Default,,0,0,0,,{karaoke_text}\n" - - with open(ass_path, "w", encoding="utf-8") as f: - f.write(header + events) - - -def format_vtt_time(seconds: float) -> str: - """Format seconds as VTT timestamp (HH:MM:SS.mmm).""" - hours = int(seconds // 3600) - minutes = int((seconds % 3600) // 60) - secs = int(seconds % 60) - millis = int((seconds % 1) * 1000) - return f"{hours:02d}:{minutes:02d}:{secs:02d}.{millis:03d}" - - -class ReelResponse(BaseModel): - success: bool - output_base64: Optional[str] = None - duration: Optional[float] = None - detail: Optional[str] = None - processing_stats: Optional[dict] = None - - -@app.get("/health") -def health_check(x_api_key: str = Header(None)): - if x_api_key != API_KEY: - raise HTTPException(status_code=401, detail="Invalid API Key") - return {"status": "healthy"} - - -@app.post("/process-reel") -async def process_reel(request: ReelRequest, x_api_key: str = Header(None)): - if x_api_key != API_KEY: - raise HTTPException(status_code=401, detail="Invalid API Key") - - start_total = time.time() - stats = { - "download_duration": 0, - "tts_duration": 0, - "stabilize_duration": 0, - "encoding_duration": 0, - "total_duration": 0, - } - - try: - job_id = str(uuid.uuid4()) - job_dir = TEMP_DIR / job_id - job_dir.mkdir() - - input_video_path = job_dir / "input.mp4" - input_audio_path = job_dir / "music.mp3" - tts_audio_path = job_dir / "tts.mp3" - tts_ass_path = job_dir / "tts.ass" - output_video_path = job_dir / "output.mp4" - - start_step = time.time() - # 1. Save Input Video - if request.video_base64: - with open(input_video_path, "wb") as f: - f.write(base64.b64decode(request.video_base64)) - elif request.video_url: - response = requests.get(request.video_url, stream=True) - response.raise_for_status() - with open(input_video_path, "wb") as f: - shutil.copyfileobj(response.raw, f) - else: - raise HTTPException(status_code=400, detail="No video source provided") - - # --- Get Video Duration for Fade Out --- - try: - video_dur_cmd = [ - "ffprobe", - "-v", - "error", - "-show_entries", - "format=duration", - "-of", - "default=noprint_wrappers=1:nokey=1", - str(input_video_path), - ] - dur_proc = subprocess.run(video_dur_cmd, stdout=subprocess.PIPE, text=True) - video_duration = float(dur_proc.stdout.strip() or 0) - except Exception as e: - print(f"⚠️ Could not measure original video duration: {e}") - video_duration = 30.0 # Fallback - - fade_duration = 2.0 - fade_start = max(0, video_duration - fade_duration) - - # Le logo doit apparaitre à 5 secondes de la fin (3 secondes avant le fondu au noir) - logo_start_time = max(0, video_duration - 5.0) - print( - f"🎬 Video Duration: {video_duration:.2f}s | Logo Start: {logo_start_time:.2f}s | Fade Out Start: {fade_start:.2f}s" - ) - - # 2. Download Music (if present) - has_music = False - if request.music_url: - try: - # Add User-Agent to avoid 403 on some CDNs - headers = {"User-Agent": "Mozilla/5.0"} - response = requests.get(request.music_url, headers=headers, stream=True) - response.raise_for_status() - with open(input_audio_path, "wb") as f: - shutil.copyfileobj(response.raw, f) - has_music = True - except Exception as e: - print(f"Failed to download music: {e}") - # We continue without music if it fails - - has_watermark = False - if request.watermark_url: - try: - # Add User-Agent to avoid 403 on some CDNs - headers = {"User-Agent": "Mozilla/5.0"} - response = requests.get( - request.watermark_url, headers=headers, stream=True - ) - response.raise_for_status() - with open(job_dir / "watermark.png", "wb") as f: - shutil.copyfileobj(response.raw, f) - has_watermark = True - except Exception as e: - print(f"Failed to download watermark: {e}") - - stats["download_duration"] = time.time() - start_step - start_step = time.time() - - # 3. Generate TTS (if enabled) - has_tts = False - tts_error = None - tts_clean_text = "" - - if request.tts_enabled and request.text: - try: - # Clean text for TTS (remove hashtags/emojis) - tts_clean_text = clean_text_for_tts(request.text) - print(f"🔊 TTS enabled. Original: '{request.text}'") - print(f"🔊 TTS cleaned: '{tts_clean_text}'") - - # Determine voice - voice = request.tts_voice - if voice == "male": - voice = "fr-FR-RemyMultilingualNeural" - elif voice == "female": - voice = "fr-FR-VivienneMultilingualNeural" - elif not voice: - voice = "fr-FR-VivienneMultilingualNeural" - # Note: Gemini voices (fr-FR-Standard-A etc.) are valid Edge voices too - print(f"🔊 Using voice: {voice}") - - if tts_clean_text: - print(f"🔊 Generating TTS audio to: {tts_audio_path}") - engine = request.tts_engine or "gemini" - - # Primary: Gemini TTS. Fallback: Edge TTS - if engine == "gemini" and request.gemini_api_key: - try: - await generate_tts_gemini( - tts_clean_text, - voice, - request.gemini_api_key, - tts_audio_path, - tts_ass_path, - display_text=clean_text_for_display(request.text), - delay=2.0, - ) - except Exception as gemini_err: - print(f"⚠️ Gemini TTS failed ({gemini_err}), falling back to Edge TTS") - await generate_tts_with_subs( - tts_clean_text, - voice, - tts_audio_path, - tts_ass_path, - display_text=clean_text_for_display(request.text), - delay=2.0, - ) - else: - await generate_tts_with_subs( - tts_clean_text, - voice, - tts_audio_path, - tts_ass_path, - display_text=clean_text_for_display(request.text), - delay=2.0, - ) - - # Verify files were created - if tts_audio_path.exists() and tts_audio_path.stat().st_size > 0: - print( - f"✅ TTS audio generated: {tts_audio_path.stat().st_size} bytes" - ) - has_tts = True - else: - print("❌ TTS audio file missing or empty!") - tts_error = "Fichier audio de la voix vide" - else: - print("⚠️ TTS text is empty after cleaning, skipping.") - except Exception as e: - import traceback - print(f"❌ Failed to generate TTS: {e}") - traceback.print_exc() - tts_error = str(e) or type(e).__name__ - - stats["tts_duration"] = time.time() - start_step - start_step = time.time() - - # 4. Build FFmpeg Command with Unified filter_complex - cmd = ["ffmpeg", "-y", "-i", str(input_video_path)] - - # --- Stability Pass 1 (if requested) --- - vidstab_filter = "" - if request.stabilize: - print("📐 Starting video stabilization (Pass 1: Detection)...") - transforms_path = job_dir / "transforms.trf" - - # Run detection pass - # Aggressive stabilization settings: - # - shakiness=10: Max sensitivity to shake - # - accuracy=15: High accuracy - # - stepsize=32: Larger search window for bigger shakes - detect_cmd = [ - "ffmpeg", - "-y", - "-i", - str(input_video_path), - "-vf", - f"vidstabdetect=stepsize=32:shakiness=10:accuracy=15:result={transforms_path}", - "-f", - "null", - "-", - ] - - detect_proc = subprocess.run( - detect_cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE - ) - - if detect_proc.returncode == 0 and transforms_path.exists(): - print( - "✅ Stabilization Pass 1 complete. Integrating Pass 2 into main filter chain." - ) - # We will add vidstabtransform to the video chain below - # smoothing=30 -> Heavy smoothing (default is 10) for handheld feel - # relative=1 -> Transforms relative to previous frame - # zoom=5 -> Fixed 5% zoom to avoid black borders from stabilization - vidstab_filter = f"vidstabtransform=input={transforms_path}:smoothing=30:relative=1:zoom=5,unsharp=5:5:1.0:5:5:0.0," - else: - print( - f"⚠️ Stabilization Pass 1 failed: {detect_proc.stderr.decode()[:500]}" - ) - - stats["stabilize_duration"] = time.time() - start_step - start_step = time.time() - - # --- Audio Checks --- - has_original_audio = False - try: - probe_cmd = [ - "ffprobe", - "-v", - "error", - "-select_streams", - "a:0", - "-show_entries", - "stream=codec_type", - "-of", - "csv=p=0", - str(input_video_path), - ] - probe_out = subprocess.check_output(probe_cmd).decode().strip() - if probe_out == "audio": - has_original_audio = True - except Exception: - pass - - # --- Inputs --- - # 0: Video (already added) - # 1: Music (optional) - # 2: TTS (optional) - - input_count = 1 - music_idx = -1 - tts_idx = -1 - - if has_music: - cmd.extend(["-i", str(input_audio_path)]) - music_idx = input_count - input_count += 1 - - if has_tts: - cmd.extend(["-i", str(tts_audio_path)]) - tts_idx = input_count - input_count += 1 - - watermark_idx = -1 - if has_watermark: - cmd.extend(["-i", str(job_dir / "watermark.png")]) - watermark_idx = input_count - input_count += 1 - - # --- Filter Complex Construction --- - fc_parts = [] - - # A. Video Chain - # Chain: [0:v] -> [stabilize] -> [scale/crop] -> [text] -> [vout] - - # 1. Stabilization (if enabled) + Scaling/Cropping - # We apply stabilization FIRST on raw video, THEN crop to 9:16 - - # Start of video chain - v_chain = "[0:v]" - - if vidstab_filter: - v_chain += vidstab_filter - # Note: vidstabtransform output is same res as input - - # Scale & Crop to Fill 1080x1920 (Vertical Reel) - # Then enhance brightness/contrast slightly for Facebook optimization - v_chain += "scale=1080:1920:force_original_aspect_ratio=increase,crop=1080:1920,eq=brightness=0.05:contrast=1.1" - - # 2. Text Overlay - if request.text and request.draw_text: - text_filter = "" - if has_tts: - # Subtitles (TikTok style) using ASS (already generated in TTS block) - print(f"🎬 Overlaying subtitles from TTS ASS: {tts_ass_path}") - ass_path_str = str(tts_ass_path).replace("\\", "/").replace(":", "\\:") - text_filter = f",subtitles='{ass_path_str}'" - else: - # Standard Text (without TTS) synchronisé via ffsubsync - print(f"🎬 Overlaying subtitles from standard text using ffsubsync...") - unsynced_srt_path = job_dir / "unsynced.srt" - synced_srt_path = job_dir / "synced.srt" - std_ass_path = job_dir / "std_text.ass" - - # Choose an audio reference - ref_audio = input_video_path - if has_music and not has_original_audio: - ref_audio = input_audio_path - - # Subtitle syncing pipeline - generate_unsynced_srt(request.text, unsynced_srt_path, total_duration=video_duration) - run_ffsubsync(ref_audio, unsynced_srt_path, synced_srt_path) - convert_srt_to_ass(synced_srt_path, std_ass_path, font_size=40, delay=0.0) - - ass_path_str = str(std_ass_path).replace("\\", "/").replace(":", "\\:") - text_filter = f",subtitles='{ass_path_str}'" - - # Combine formatting + text - v_chain += text_filter - - if has_watermark: - if request.store_name and request.enable_ending_effect: - # Ouro Party Mode + Persistent bottom right - - # We need two scaled versions of the logo - # [wm_small]: Bottom right persistent logo - fc_parts.append(f"[{watermark_idx}:v]scale=200:-1,split=2[wm_small_base][wm_large_base]") - fc_parts.append(f"[wm_large_base]scale=-1:300[wm_large]") - - # 1. Place small logo in bottom right until logo_start_time (5s before the end) - v_chain += f"[v_pre_small];[v_pre_small][wm_small_base]overlay=W-w-20:H-h-20:enable='between(t,0,{logo_start_time})'" - - # 2. Place large logo in the center, and fading it IN during the last 5 seconds - v_chain += f"[v_pre_large];[v_pre_large][wm_large]overlay=(W-w)/2:(H-h)/2-100:enable='between(t,{logo_start_time},{video_duration})'" - - # 3. Drawing the Store Name below the logo using ASS subtitles - outro_ass_path = job_dir / "outro.ass" - generate_outro_ass( - request.store_name, outro_ass_path, logo_start_time, video_duration - ) - ass_path_str_2 = ( - str(outro_ass_path).replace("\\", "/").replace(":", "\\:") - ) - v_chain += f",subtitles='{ass_path_str_2}'" - else: - # Normal watermark (bottom right) - fc_parts.append(f"[{watermark_idx}:v]scale=200:-1[wm]") - v_chain += "[v_pre_wm];[v_pre_wm][wm]overlay=W-w-20:H-h-20" - - # Add Video Fade Out - if request.enable_ending_effect: - v_chain += f",fade=t=out:st={fade_start}:d={fade_duration}" - - - # End of video chain - v_chain += "[vout]" - fc_parts.append(v_chain) - - # B. Audio Chain - audio_mapped = False - - inputs_for_mix = 0 - audio_mix_str = "" - - # Strategy: - # If no music and no TTS -> Copy original audio (if exists) or silent - # If music or TTS -> Mix everything - - if has_music or has_tts: - # When music or TTS is used, we REMOVE the original video audio - # and only mix the new audio sources (music + TTS) - # Original audio is intentionally excluded to avoid background noise/voices - - if has_music: - # Adjust volume - fc_parts.append( - f"[{music_idx}:a]volume={request.music_volume}[a_music]" - ) - audio_mix_str += "[a_music]" - inputs_for_mix += 1 - - if has_tts: - # TTS louder and delayed by 2 seconds (2s) on all channels - fc_parts.append(f"[{tts_idx}:a]adelay=2s:all=1,volume=1.5[a_tts]") - audio_mix_str += "[a_tts]" - inputs_for_mix += 1 - - # Mix - if inputs_for_mix > 0: - fc_parts.append( - f"{audio_mix_str}amix=inputs={inputs_for_mix}:duration=first:dropout_transition=2:normalize=0[amixout]" - ) - fc_parts.append( - f"[amixout]afade=t=out:st={fade_start}:d={fade_duration}[aout]" - ) - audio_mapped = True - else: - # No external audio added - # To prevent FFmpeg crashes with certain MP4 original audio codecs, - # we bypass the afade filter entirely and just map 0:a directly. - audio_mapped = False - - - # Apply Filter Complex - cmd.extend(["-filter_complex", ";".join(fc_parts)]) - - # Maps - cmd.extend(["-map", "[vout]"]) # Map processed video - - if audio_mapped: - cmd.extend(["-map", "[aout]"]) # Map mixed audio - elif has_original_audio: - cmd.extend(["-map", "0:a"]) # Map original audio directly - - # Cut EXACTLY at video length (better than -shortest which can cause issues with amix) - cmd.extend(["-t", str(video_duration)]) - - # Quality settings - cmd.extend( - [ - "-c:v", - "libx264", - "-profile:v", - "high", - "-r", - "30", - "-preset", - "slow", - "-level", - "4.1", - "-crf", - "18", - "-b:v", - "10M", - "-maxrate", - "12M", - "-bufsize", - "20M", - "-c:a", - "aac", - "-b:a", - "128k", - "-pix_fmt", - "yuv420p", - "-movflags", - "+faststart", - ] - ) - - cmd.append(str(output_video_path)) - print(f"🚀 Executing FFmpeg command: {' '.join(cmd)}") - - # execute - process = subprocess.run(cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE) - - # Log detailed output on failure OR success for debugging font issues - if process.returncode != 0: - print(f"❌ FFmpeg failed. Stderr:\n{process.stderr.decode()}") - else: - # Check stderr for font warnings even on success - stderr_last_lines = "\n".join(process.stderr.decode().splitlines()[-20:]) - print(f"✅ FFmpeg executed. Stderr (last 20 lines):\n{stderr_last_lines}") - - stats["encoding_duration"] = time.time() - start_step - stats["total_duration"] = time.time() - start_total - - if process.returncode != 0: - raise Exception(f"FFmpeg encoding failed: {process.stderr.decode()}") - - # 4. Get Duration (ffprobe) - duration_cmd = [ - "ffprobe", - "-v", - "error", - "-show_entries", - "format=duration", - "-of", - "default=noprint_wrappers=1:nokey=1", - str(output_video_path), - ] - dur_proc = subprocess.run(duration_cmd, stdout=subprocess.PIPE) - duration = float(dur_proc.stdout.decode().strip() or 0) - - # 5. Read Output - with open(output_video_path, "rb") as f: - out_bytes = f.read() - out_b64 = base64.b64encode(out_bytes).decode("utf-8") - - # Cleanup - shutil.rmtree(job_dir) - - print(f"📊 Processing Stats: {stats}") - - return { - "success": True, - "output_base64": out_b64, - "duration": duration, - "processing_stats": stats, - "tts_error": tts_error, - } - - except Exception as e: - if "job_dir" in locals(): - shutil.rmtree(job_dir, ignore_errors=True) - return {"success": False, "detail": str(e)} - - -@app.post("/preview-tts") -async def preview_tts(request: ReelRequest, x_api_key: str = Header(None)): - if x_api_key != API_KEY: - raise HTTPException(status_code=401, detail="Invalid API Key") - - try: - job_id = str(uuid.uuid4()) - job_dir = TEMP_DIR / job_id - job_dir.mkdir() - - tts_audio_path = job_dir / "preview.mp3" - tts_srt_path = job_dir / "preview.srt" # Consistent with rest of app - - if not request.text: - raise HTTPException(status_code=400, detail="Text required for preview") - - clean_text = clean_text_for_tts(request.text) - - # Determine voice - voice = request.tts_voice - if voice == "male": - voice = "fr-FR-RemyMultilingualNeural" - elif voice == "female": - voice = "fr-FR-VivienneMultilingualNeural" - elif not voice: - voice = "fr-FR-VivienneMultilingualNeural" - # Note: Gemini voices (fr-FR-Standard-A etc.) are valid Edge voices too, - # so we don't check for "Neural" - let edge_tts handle voice resolution - - engine = request.tts_engine or "gemini" - - # Primary: Gemini TTS. Fallback: Edge TTS - if engine == "gemini" and request.gemini_api_key: - try: - await generate_tts_gemini( - clean_text, - voice, - request.gemini_api_key, - tts_audio_path, - tts_srt_path, - display_text=clean_text_for_display(request.text), - ) - except Exception as gemini_err: - print(f"⚠️ Gemini TTS failed ({gemini_err}), falling back to Edge TTS") - await generate_tts_with_subs( - clean_text, - voice, - tts_audio_path, - tts_srt_path, - display_text=clean_text_for_display(request.text), - ) - else: - await generate_tts_with_subs( - clean_text, - voice, - tts_audio_path, - tts_srt_path, - display_text=clean_text_for_display(request.text), - ) - - if not tts_audio_path.exists(): - raise Exception("TTS generation failed (file missing)") - - with open(tts_audio_path, "rb") as f: - audio_bytes = f.read() - audio_b64 = base64.b64encode(audio_bytes).decode("utf-8") - - shutil.rmtree(job_dir) - return {"success": True, "audio_base64": audio_b64} - - except Exception as e: - if "job_dir" in locals(): - shutil.rmtree(job_dir, ignore_errors=True) - return {"success": False, "detail": str(e)} +from app.api import app +__all__ = ["app"] if __name__ == "__main__": + import uvicorn + uvicorn.run(app, host="0.0.0.0", port=8000) diff --git a/ffmpeg-service/requirements-dev.txt b/ffmpeg-service/requirements-dev.txt new file mode 100644 index 0000000..d42aa18 --- /dev/null +++ b/ffmpeg-service/requirements-dev.txt @@ -0,0 +1,3 @@ +-r requirements.txt +pytest==8.3.5 +ruff==0.11.8 diff --git a/ffmpeg-service/requirements.txt b/ffmpeg-service/requirements.txt index 997b044..7b030b5 100644 --- a/ffmpeg-service/requirements.txt +++ b/ffmpeg-service/requirements.txt @@ -1,9 +1,7 @@ -fastapi==0.109.0 -uvicorn==0.27.0 -python-multipart==0.0.6 -requests==2.31.0 -pydantic==2.6.0 -edge-tts==6.1.12 -emoji -ffsubsync==0.4.26 -httpx>=0.25.0 +fastapi==0.115.12 +uvicorn[standard]==0.34.2 +pydantic==2.11.4 +httpx==0.28.1 +edge-tts==7.2.8 +emoji==2.14.1 +faster-whisper==1.2.1 diff --git a/ffmpeg-service/ruff.toml b/ffmpeg-service/ruff.toml new file mode 100644 index 0000000..46ea7a5 --- /dev/null +++ b/ffmpeg-service/ruff.toml @@ -0,0 +1,9 @@ +line-length = 110 +target-version = "py311" + +[lint] +select = ["E", "F", "W", "I", "B", "UP"] +ignore = [ + "B008", # Depends() dans les signatures FastAPI + "B905", # zip() sur des séquences dont les longueurs sont liées par construction +] diff --git a/ffmpeg-service/test.ass b/ffmpeg-service/test.ass deleted file mode 100644 index 0616572..0000000 --- a/ffmpeg-service/test.ass +++ /dev/null @@ -1,14 +0,0 @@ -[Script Info] -ScriptType: v4.00+ -PlayResX: 1080 -PlayResY: 1920 -ScaledBorderAndShadow: yes - -[V4+ Styles] -Format: Name, Fontname, Fontsize, PrimaryColour, SecondaryColour, OutlineColour, BackColour, Bold, Italic, Underline, StrikeOut, ScaleX, ScaleY, Spacing, Angle, BorderStyle, Outline, Shadow, Alignment, MarginL, MarginR, MarginV, Encoding -Style: Default,Sans,65,&H0000FFFF,&H00FFFFFF,&H00000000,&H80000000,-1,0,0,0,100,100,0,0,1,4,2,5,50,50,0,1 - -[Events] -Format: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text -Dialogue: 0,0:00:02.00,0:00:02.71,Default,,0,0,0,,{\kf31}Test {\kf31}test {\kf10}1 -Dialogue: 0,0:00:02.71,0:00:02.87,Default,,0,0,0,,{\kf10}2 {\kf10}3 diff --git a/ffmpeg-service/test2.ass b/ffmpeg-service/test2.ass deleted file mode 100644 index 979b50d..0000000 --- a/ffmpeg-service/test2.ass +++ /dev/null @@ -1,13 +0,0 @@ -[Script Info] -ScriptType: v4.00+ -PlayResX: 1080 -PlayResY: 1920 -ScaledBorderAndShadow: yes - -[V4+ Styles] -Format: Name, Fontname, Fontsize, PrimaryColour, SecondaryColour, OutlineColour, BackColour, Bold, Italic, Underline, StrikeOut, ScaleX, ScaleY, Spacing, Angle, BorderStyle, Outline, Shadow, Alignment, MarginL, MarginR, MarginV, Encoding -Style: Default,Sans,65,&H00FFFFFF,&H000000FF,&H00000000,&H80000000,-1,0,0,0,100,100,0,0,1,4,2,5,50,50,0,1 - -[Events] -Format: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text -Dialogue: 0,0:00:02.00,0:00:04.00,Default,,0,0,0,,Hello World diff --git a/ffmpeg-service/tests/__init__.py b/ffmpeg-service/tests/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/ffmpeg-service/tests/conftest.py b/ffmpeg-service/tests/conftest.py new file mode 100644 index 0000000..a25040a --- /dev/null +++ b/ffmpeg-service/tests/conftest.py @@ -0,0 +1,3 @@ +import os + +os.environ.setdefault("API_KEY", "test-key") diff --git a/ffmpeg-service/tests/test_align.py b/ffmpeg-service/tests/test_align.py new file mode 100644 index 0000000..c2ba1c2 --- /dev/null +++ b/ffmpeg-service/tests/test_align.py @@ -0,0 +1,31 @@ +from app.align import Word, align_words, display_tokens, normalize + + +def test_normalize_strips_accents_and_punctuation(): + assert normalize("Été,") == "ete" + assert normalize("«") == "" + + +def test_display_tokens_attach_isolated_punctuation(): + assert display_tokens("Bonjour ! Venez vite .") == ["Bonjour!", "Venez", "vite."] + + +def test_align_uses_engine_timings_and_keeps_display_text(): + spoken = [Word("Bonjour", 0.05, 0.6), Word("découvrez", 0.99, 1.35), Word("nos", 1.35, 1.44)] + words = align_words("Bonjour, découvrez nos", spoken) + assert [w.text for w in words] == ["Bonjour,", "découvrez", "nos"] + assert words[1].start == 0.99 + + +def test_unmatched_words_are_interpolated_between_neighbours(): + spoken = [Word("les", 0.0, 0.2), Word("prix", 1.0, 1.3)] + words = align_words("les super promos prix", spoken) + assert [w.text for w in words] == ["les", "super", "promos", "prix"] + assert 0.2 <= words[1].start < words[2].start < 1.0 + assert all(b.start > a.start for a, b in zip(words, words[1:])) + + +def test_without_spoken_words_text_is_spread_over_duration(): + words = align_words("un deux trois", [], total_duration=3.0) + assert words[0].start == 0.0 + assert abs(words[-1].end - 3.0) < 1e-6 diff --git a/ffmpeg-service/tests/test_render.py b/ffmpeg-service/tests/test_render.py new file mode 100644 index 0000000..94b5bd8 --- /dev/null +++ b/ffmpeg-service/tests/test_render.py @@ -0,0 +1,77 @@ +from pathlib import Path + +from app.render import RenderPlan, build_command + + +def _graph(cmd): + return cmd[cmd.index("-filter_complex") + 1] + + +def test_voice_longer_than_video_extends_with_frozen_frame(): + plan = RenderPlan( + video=Path("in.mp4"), + video_duration=5.0, + output=Path("out.mp4"), + voice=Path("v.wav"), + voice_duration=10.0, + ) + assert plan.total_duration == 12.8 + cmd = build_command(plan) + assert "tpad=stop_mode=clone:stop_duration=7.800" in _graph(cmd) + assert cmd[cmd.index("-t") + 1] == "12.800" + + +def test_music_is_looped_and_ducked_under_voice(): + plan = RenderPlan( + video=Path("in.mp4"), + video_duration=20.0, + output=Path("out.mp4"), + music=Path("m.mp3"), + voice=Path("v.wav"), + voice_duration=5.0, + ) + cmd = build_command(plan) + assert cmd[cmd.index("m.mp3") - 3 : cmd.index("m.mp3")] == ["-stream_loop", "-1", "-i"] + graph = _graph(cmd) + assert "sidechaincompress" in graph + assert "amix=inputs=2:duration=longest" in graph + assert "loudnorm=I=-14" in graph + + +def test_no_brightness_hack_and_hdr_tonemapped(): + plan = RenderPlan(video=Path("in.mp4"), video_duration=5.0, output=Path("out.mp4"), is_hdr=True) + graph = _graph(build_command(plan)) + assert "eq=" not in graph + assert "tonemap=tonemap=hable" in graph + assert "flags=lanczos" in graph + + +def test_original_audio_kept_when_nothing_added(): + plan = RenderPlan( + video=Path("in.mp4"), video_duration=5.0, output=Path("out.mp4"), keep_original_audio=True + ) + cmd = build_command(plan) + assert cmd[cmd.index("-map", cmd.index("[vout]")) + 1] == "0:a:0" + + +def test_encoding_targets_social_networks(): + cmd = build_command(RenderPlan(video=Path("in.mp4"), video_duration=5.0, output=Path("out.mp4"))) + joined = " ".join(cmd) + assert "-crf 19" in joined and "-b:v" not in joined + assert "-ar 48000" in joined and "-b:a 192k" in joined + assert "-colorspace bt709" in joined + + +def test_big_logo_waits_for_the_end_of_the_voice(): + plan = RenderPlan( + video=Path("in.mp4"), + video_duration=6.0, + output=Path("out.mp4"), + voice=Path("v.wav"), + voice_duration=4.8, + watermark=Path("logo.png"), + outro=Path("o.ass"), + ) + assert plan.logo_start == 6.8 + assert plan.total_duration == 9.3 + assert "gte(t,6.800)" in _graph(build_command(plan)) diff --git a/ffmpeg-service/tests/test_subtitles.py b/ffmpeg-service/tests/test_subtitles.py new file mode 100644 index 0000000..359cb19 --- /dev/null +++ b/ffmpeg-service/tests/test_subtitles.py @@ -0,0 +1,23 @@ +from app.align import Word +from app.subtitles import ass_time, group_lines, write_captions + + +def test_ass_time_format(): + assert ass_time(0) == "0:00:00.00" + assert ass_time(61.234) == "0:01:01.23" + + +def test_lines_break_on_punctuation_and_length(): + words = [Word(t, i, i + 0.5) for i, t in enumerate("Bonjour, venez découvrir nos nouveautés".split())] + lines = group_lines(words) + assert [w.text for w in lines[0]] == ["Bonjour,"] + assert all(len(line) <= 3 for line in lines) + + +def test_captions_are_offset_and_escape_braces(tmp_path): + path = tmp_path / "c.ass" + write_captions([Word("{prix}", 0.0, 0.5), Word("fous", 0.5, 1.0)], path, offset=2.0, font_size=80) + content = path.read_text() + assert "Dialogue: 0,0:00:02.00," in content + assert "{prix}" not in content.split("[Events]")[1].replace("{\\", "") + assert "(prix)" in content diff --git a/ffmpeg-service/tests/test_voices.py b/ffmpeg-service/tests/test_voices.py new file mode 100644 index 0000000..44e85b3 --- /dev/null +++ b/ffmpeg-service/tests/test_voices.py @@ -0,0 +1,23 @@ +from app.tts.gemini import build_prompt +from app.voices import GEMINI_VOICES, edge_fallbacks, resolve_edge_voice, resolve_gemini_voice + + +def test_thirty_gemini_voices(): + assert len(GEMINI_VOICES) == 30 + + +def test_legacy_gemini_identifiers_still_resolve(): + assert resolve_gemini_voice("fr-FR-Standard-B").id == "Charon" + assert resolve_gemini_voice("fr-FR-Standard-A").id == "Kore" + assert resolve_gemini_voice("puck").id == "Puck" + + +def test_edge_fallbacks_are_french_and_same_gender(): + voice = resolve_edge_voice("Fenrir") # voix Gemini masculine + assert voice.gender == "male" + assert all(v.id.startswith("fr-FR") and v.gender == "male" for v in edge_fallbacks(voice)) + + +def test_style_prompt_precedes_text(): + assert build_prompt("Salut", "neutral") == "Salut" + assert build_prompt("Salut", "dynamic").endswith(":\nSalut") diff --git a/server/routes/reels.ts b/server/routes/reels.ts index 4d7b1c4..ddd9034 100644 --- a/server/routes/reels.ts +++ b/server/routes/reels.ts @@ -5,14 +5,13 @@ import { Router, Request, Response } from 'express'; import type { User } from '@shared/schema'; import { storage } from '../storage'; -import { ffmpegService } from '../services/ffmpeg'; -import { resolveInternalUrl } from '../services/minio'; -import { videoReelParamsSchema, type VideoReelParams } from '@shared/reel'; +import { ffmpegService, FFmpegServiceError } from '../services/ffmpeg'; +import { ttsPreviewSchema, videoReelParamsSchema, type VideoReelParams } from '@shared/reel'; import { enqueueReelJob, countActiveReelJobs } from '../services/reels/queue'; -import { resolveGeminiApiKey, resolveLogoPath, resolveMusicUrl, resolveStoreName } from '../services/reels/assets'; +import { resolveGeminiApiKey, resolveStoreName } from '../services/reels/assets'; import { openRouterService, describeGenerationError } from '../services/openrouter'; -import { ttsSyncService } from '../services/ttsSync'; +import { estimateVoiceTiming } from '../services/ttsSync'; /** Piste musicale telle qu'attendue par le client. */ interface MusicTrack { id: string; @@ -270,105 +269,37 @@ reelsRouter.post('/reels/generate-text', async (req: Request, res: Response) => } }); -/** - * Prévisualiser un Reel (traitement sans publication) - * POST /api/reels/preview - */ -reelsRouter.post('/reels/preview', async (req: Request, res: Response) => { - try { - const user = req.user as User; - const { - videoMediaId, - musicTrackId, - musicUrl, - overlayText, - ttsEnabled, - ttsVoice, - ttsEngine, - wordDuration = 0.6, - fontSize = 64, - musicVolume = 0.25, - drawText = true, - stabilize = false, - enableEndingEffect = true, - } = req.body; - - // Récupérer le média vidéo - const media = await storage.getMediaById(videoMediaId); - if (!media) { - return res.status(404).json({ error: 'Vidéo non trouvée' }); - } - - if (media.type !== 'video') { - return res.status(400).json({ error: 'Le média doit être une vidéo' }); - } - - const [finalMusicUrl, logoPath, geminiApiKey] = await Promise.all([ - resolveMusicUrl(musicTrackId, musicUrl), - resolveLogoPath(), - resolveGeminiApiKey(ttsEngine), - ]); - const watermarkUrl = logoPath ? resolveInternalUrl(logoPath) : undefined; - - const finalWordDuration = wordDuration; - - // Traiter la vidéo via FFmpeg - const result = await ffmpegService.processReelFromUrl(resolveInternalUrl(media.originalUrl), { - text: overlayText, - musicUrl: finalMusicUrl, - ttsEnabled, - ttsVoice, - ttsEngine, - geminiApiKey, - wordDuration: finalWordDuration, - fontSize, - musicVolume, - drawText, - stabilize, - watermarkUrl, - enableEndingEffect, - }); - - if (!result.success) { - return res.status(500).json({ error: result.error || 'Erreur de traitement vidéo' }); - } - - // Retourner la vidéo en base64 pour prévisualisation - res.json({ - success: true, - videoBase64: result.videoBase64, - duration: result.duration, - }); - } catch (error) { - console.error('❌ Error previewing Reel:', error); - res.status(500).json({ error: 'Erreur lors de la prévisualisation du Reel' }); - } -}); - /** * Prévisualiser la voix TTS * POST /api/reels/tts-preview */ reelsRouter.post('/reels/tts-preview', async (req: Request, res: Response) => { + const parsed = ttsPreviewSchema.safeParse(req.body); + if (!parsed.success) { + return res.status(400).json({ error: parsed.error.issues[0]?.message ?? 'Paramètres invalides' }); + } + const { text, ttsVoice, ttsEngine, ttsStyle } = parsed.data; + try { - const user = req.user as User; - const { text, ttsVoice, ttsEngine } = req.body; - - if (!text) { - return res.status(400).json({ error: 'Texte requis' }); - } - - const geminiApiKey = await resolveGeminiApiKey(ttsEngine); - const result = await ffmpegService.previewTTS(text, ttsVoice, ttsEngine, geminiApiKey); - - if (!result.success) { - return res.status(500).json({ error: result.error || 'Erreur de génération TTS' }); - } - - res.json({ success: true, audioBase64: result.audioBase64 }); + const preview = await ffmpegService.previewVoice(text, { + voice: ttsVoice, + engine: ttsEngine, + style: ttsStyle, + geminiApiKey: await resolveGeminiApiKey(ttsEngine), + }); + res.json({ + success: true, + audioBase64: preview.audio.toString('base64'), + duration: preview.duration, + words: preview.words, + engine: preview.engine, + voice: preview.voice, + warnings: preview.warnings, + }); } catch (error) { console.error('❌ Error generating TTS preview:', error); - res.status(500).json({ error: 'Erreur lors de la génération de la voix' }); + const message = error instanceof Error ? error.message : 'Erreur lors de la génération de la voix'; + res.status(error instanceof FFmpegServiceError && error.status === 400 ? 400 : 502).json({ error: message }); } }); @@ -378,13 +309,12 @@ reelsRouter.post('/reels/tts-preview', async (req: Request, res: Response) => { */ reelsRouter.post('/reels/sync-info', async (req: Request, res: Response) => { try { - const { text, ttsVoice, ttsEngine } = req.body; - if (!text || !ttsVoice) { - return res.status(400).json({ error: 'Texte et voix requis' }); + const { text } = req.body; + if (!text) { + return res.status(400).json({ error: 'Texte requis' }); } - const geminiApiKey = await resolveGeminiApiKey(ttsEngine); - const sync = await ttsSyncService.calculateSyncTiming(text, ttsVoice, ttsEngine, geminiApiKey); - res.json(sync); + // Estimation locale : ne déclenche aucune synthèse (payante avec Gemini) + res.json(estimateVoiceTiming(text)); } catch (error) { console.error('❌ Error calculating sync:', error); res.status(500).json({ error: 'Erreur de calcul de synchronisation' }); diff --git a/server/routes/remotion.ts b/server/routes/remotion.ts index 1fc3205..25958ba 100644 --- a/server/routes/remotion.ts +++ b/server/routes/remotion.ts @@ -70,8 +70,10 @@ remotionRouter.post("/render", upload.fields([{ name: "images", maxCount: 4 }, { overlayText: req.body.overlayText, musicUrl: musicFile ? `/uploads/temp/${path.basename(musicFile.path)}` : req.body.musicTrackUrl, musicVolume: Number.isFinite(musicVolume) ? musicVolume : undefined, + ttsEnabled: req.body.ttsEnabled !== "false", ttsEngine: req.body.ttsEngine || undefined, ttsVoice: req.body.ttsVoice, + ttsStyle: req.body.ttsStyle || undefined, storeName: await resolveStoreName(user.id, req.body.selectedPageId), tempFiles: uploaded.map((f) => f.path), }); diff --git a/server/services/ffmpeg.ts b/server/services/ffmpeg.ts index 94a87a3..289a78b 100644 --- a/server/services/ffmpeg.ts +++ b/server/services/ffmpeg.ts @@ -1,214 +1,120 @@ /** - * FFmpeg Docker API Service - * - * Intégration avec l'API FFmpeg Docker locale pour le traitement vidéo des Reels. - * L'API attend une vidéo en base64 et retourne la vidéo traitée en base64. + * Client du service FFmpeg (conteneur Python `ffmpeg-service`). + * + * Le service rend le Reel puis expose le MP4 en téléchargement + * (GET /files/{job}/output.mp4) : la vidéo ne transite plus en base64 dans du + * JSON, qui gonflait sa taille d'un tiers et la gardait entière en mémoire. */ -interface FFmpegReelRequest { - video_base64?: string; // Vidéo source en base64 - video_url?: string; // OU URL de la vidéo source - text?: string; // Texte overlay style TikTok - music_id?: string; // ID de la musique (catalogue FFmpeg) - music_url?: string; // OU URL directe de la musique - tts_enabled?: boolean; // Activation du TTS - tts_voice?: string; // Voix TTS (ex: fr-FR-VivienneNeural) - tts_engine?: string; // Moteur TTS: "edge" ou "gemini" - gemini_api_key?: string; // Clé API Google Gemini pour TTS - word_duration?: number; // Durée par mot (default: 0.6s) - font_size?: number; // Taille police (default: 24) - music_volume?: number; // Volume musique (default: 0.25) - draw_text?: boolean; // Dessiner le texte sur la vidéo (default: true) - stabilize?: boolean; // Stabilisation vidéo via vidstab (default: false) - watermark_url?: string; // URL du logo - store_name?: string; // Nom du magasin pour l'outro - enable_ending_effect?: boolean; // Activer l'effet de fin (logo+fondu) -} - -interface FFmpegReelResponse { - success: boolean; - output_base64?: string; - duration?: number; - detail?: string; - tts_error?: string; -} +import type { TtsEngine, TtsStyle } from '@shared/voices'; /** Un rendu long (stabilisation + encodage) peut dépasser plusieurs minutes. */ -const PROCESS_TIMEOUT_MS = 15 * 60_000; -const TTS_TIMEOUT_MS = 2 * 60_000; +const PROCESS_TIMEOUT_MS = 20 * 60_000; +const DOWNLOAD_TIMEOUT_MS = 5 * 60_000; +const TTS_TIMEOUT_MS = 3 * 60_000; const HEALTH_TIMEOUT_MS = 5_000; +export interface ReelRenderOptions { + text?: string; + musicUrl?: string; + ttsEnabled?: boolean; + ttsVoice?: string; + ttsEngine?: TtsEngine; + ttsStyle?: TtsStyle; + geminiApiKey?: string; + fontSize?: number; + musicVolume?: number; + drawText?: boolean; + stabilize?: boolean; + watermarkUrl?: string; + storeName?: string; + enableEndingEffect?: boolean; +} + +export interface ReelRenderResult { + video: Buffer; + duration: number; + ttsEngine?: string | null; + ttsVoice?: string | null; + warnings: string[]; +} + +export interface TimedWord { + text: string; + start: number; + end: number; +} + +export interface VoicePreview { + audio: Buffer; + duration: number; + words: TimedWord[]; + engine: string; + voice: string; + warnings: string[]; +} + interface FFmpegConfig { apiUrl: string; apiKey: string; } +/** Erreur renvoyée par le service, avec son message lisible. */ +export class FFmpegServiceError extends Error { + constructor(message: string, readonly status?: number) { + super(message); + this.name = 'FFmpegServiceError'; + } +} + export class FFmpegService { private config: FFmpegConfig | null = null; - /** - * Configure le service avec l'URL et la clé API - */ configure(apiUrl: string, apiKey: string): void { - this.config = { apiUrl, apiKey }; - console.log('🎬 FFmpeg Service configured:', apiUrl); + this.config = { apiUrl: apiUrl.replace(/\/$/, ''), apiKey }; + console.log('🎬 FFmpeg Service configured:', this.config.apiUrl); } - /** - * Vérifie que le service est configuré - */ private ensureConfigured(): FFmpegConfig { if (!this.config) { - throw new Error('FFmpeg Service not configured. Call configure() first.'); + throw new FFmpegServiceError("Le service FFmpeg n'est pas configuré (FFMPEG_API_URL / FFMPEG_API_KEY)."); } return this.config; } - /** - * Traite une vidéo pour créer un Reel avec musique et texte overlay - * - * @param videoBase64 - Vidéo source encodée en base64 - * @param options - Options de traitement (texte, musique, etc.) - * @returns Vidéo traitée en base64 - */ - async processReelVideo( - videoBase64: string, - options: { - text?: string; - musicId?: string; - musicUrl?: string; - ttsEnabled?: boolean; - ttsVoice?: string; - ttsEngine?: string; - geminiApiKey?: string; - wordDuration?: number; - fontSize?: number; - musicVolume?: number; - drawText?: boolean; - stabilize?: boolean; - watermarkUrl?: string; - storeName?: string; - enableEndingEffect?: boolean; - } = {} - ): Promise<{ success: boolean; videoBase64?: string; duration?: number; error?: string }> { + private async call(path: string, init: RequestInit & { timeoutMs: number }): Promise { const config = this.ensureConfigured(); - - const requestBody: FFmpegReelRequest = { - video_base64: videoBase64, - text: options.text, - music_id: options.musicId, - music_url: options.musicUrl, - tts_enabled: options.ttsEnabled, - tts_voice: options.ttsVoice, - tts_engine: options.ttsEngine, - gemini_api_key: options.geminiApiKey, - word_duration: options.wordDuration ?? 0.6, - font_size: options.fontSize ?? 64, - music_volume: options.musicVolume ?? 0.25, - draw_text: options.drawText ?? true, - stabilize: options.stabilize ?? false, - watermark_url: options.watermarkUrl, - store_name: options.storeName, - enable_ending_effect: options.enableEndingEffect ?? true, - }; - - // Remove undefined values - Object.keys(requestBody).forEach(key => { - if (requestBody[key as keyof FFmpegReelRequest] === undefined) { - delete requestBody[key as keyof FFmpegReelRequest]; - } + const { timeoutMs, headers, ...rest } = init; + const response = await fetch(`${config.apiUrl}${path}`, { + ...rest, + headers: { 'X-API-Key': config.apiKey, ...headers }, + signal: AbortSignal.timeout(timeoutMs), }); - - console.log('🎬 Processing Reel video:', { - hasVideo: !!videoBase64, - hasText: !!options.text, - hasMusicId: !!options.musicId, - hasMusicUrl: !!options.musicUrl, - hasTTS: options.ttsEnabled, - drawText: options.drawText, - }); - - try { - const response = await fetch(`${config.apiUrl}/process-reel`, { - method: 'POST', - headers: { - 'Content-Type': 'application/json', - 'X-API-Key': config.apiKey, - }, - body: JSON.stringify(requestBody), - signal: AbortSignal.timeout(PROCESS_TIMEOUT_MS), - }); - - if (!response.ok) { - const errorText = await response.text(); - console.error('❌ FFmpeg API error:', response.status, errorText); - return { - success: false, - error: `FFmpeg API error: ${response.status} - ${errorText}`, - }; - } - - const data = await response.json() as FFmpegReelResponse; - - if (!data.success) { - console.error('❌ FFmpeg processing failed:', data.detail); - return { - success: false, - error: data.detail || 'Unknown FFmpeg processing error', - }; - } - - console.log('✅ Reel video processed successfully, duration:', data.duration); - return { - success: true, - videoBase64: data.output_base64, - duration: data.duration, - }; - - } catch (error) { - console.error('❌ FFmpeg Service error:', error); - return { - success: false, - error: error instanceof Error ? error.message : 'Unknown error', - }; + if (!response.ok) { + const body = await response.text(); + let detail = body; + try { + detail = JSON.parse(body).detail ?? body; + } catch { /* corps non JSON */ } + throw new FFmpegServiceError(String(detail).slice(0, 2000), response.status); } + return response; } /** - * Traite une vidéo depuis une URL (télécharge, traite, retourne base64) + * Rend un Reel à partir de l'URL d'une vidéo et renvoie le MP4 produit. + * Lève FFmpegServiceError en cas d'échec (voix comprise). */ - async processReelFromUrl( - videoUrl: string, - options: { - text?: string; - musicId?: string; - musicUrl?: string; - ttsEnabled?: boolean; - ttsVoice?: string; - ttsEngine?: string; - geminiApiKey?: string; - wordDuration?: number; - fontSize?: number; - musicVolume?: number; - drawText?: boolean; - stabilize?: boolean; - watermarkUrl?: string; - storeName?: string; - enableEndingEffect?: boolean; - } = {} - ): Promise<{ success: boolean; videoBase64?: string; duration?: number; error?: string; ttsError?: string }> { - const config = this.ensureConfigured(); - - const requestBody: FFmpegReelRequest = { + async renderReel(videoUrl: string, options: ReelRenderOptions = {}): Promise { + const body = { video_url: videoUrl, text: options.text, - music_id: options.musicId, music_url: options.musicUrl, - tts_enabled: options.ttsEnabled, + tts_enabled: options.ttsEnabled ?? false, tts_voice: options.ttsVoice, tts_engine: options.ttsEngine, + tts_style: options.ttsStyle, gemini_api_key: options.geminiApiKey, - word_duration: options.wordDuration ?? 0.6, font_size: options.fontSize ?? 64, music_volume: options.musicVolume ?? 0.25, draw_text: options.drawText ?? true, @@ -218,139 +124,88 @@ export class FFmpegService { enable_ending_effect: options.enableEndingEffect ?? true, }; - // Remove undefined values - Object.keys(requestBody).forEach(key => { - if (requestBody[key as keyof FFmpegReelRequest] === undefined) { - delete requestBody[key as keyof FFmpegReelRequest]; - } - }); - - console.log('🎬 Processing Reel from URL:', { + console.log('🎬 Rendu du Reel :', { videoUrl, - hasText: !!options.text, - textLength: options.text?.length || 0, - hasMusicId: !!options.musicId, - hasMusicUrl: !!options.musicUrl, - ttsEnabled: options.ttsEnabled, - drawText: options.drawText, + textLength: options.text?.length ?? 0, + music: !!options.musicUrl, + tts: options.ttsEnabled ? `${options.ttsEngine}/${options.ttsVoice}/${options.ttsStyle ?? 'neutral'}` : false, }); - const debugBody = { ...requestBody }; - console.log('📤 Sending to FFmpeg API:', JSON.stringify({ ...debugBody, text: debugBody.text ? `[${debugBody.text.length} chars]` : undefined })); + const response = await this.call('/process-reel', { + method: 'POST', + headers: { 'Content-Type': 'application/json' }, + body: JSON.stringify(body), + timeoutMs: PROCESS_TIMEOUT_MS, + }); + const data = await response.json() as { + job_id: string; + output_path: string; + duration: number; + tts_engine?: string | null; + tts_voice?: string | null; + warnings?: string[]; + }; try { - const response = await fetch(`${config.apiUrl}/process-reel`, { - method: 'POST', - headers: { - 'Content-Type': 'application/json', - 'X-API-Key': config.apiKey, - }, - body: JSON.stringify(requestBody), - signal: AbortSignal.timeout(PROCESS_TIMEOUT_MS), - }); - - if (!response.ok) { - const errorText = await response.text(); - console.error('❌ FFmpeg API error:', response.status, errorText); - return { - success: false, - error: `FFmpeg API error: ${response.status} - ${errorText}`, - }; - } - - const data = await response.json() as FFmpegReelResponse; - - if (!data.success) { - console.error('❌ FFmpeg processing failed:', data.detail); - return { - success: false, - error: data.detail || 'Unknown FFmpeg processing error', - }; - } - - if (data.tts_error) { - console.error('❌ TTS failed in Python service:', data.tts_error); - } - console.log('✅ Reel video processed successfully from URL'); + const file = await this.call(data.output_path, { method: 'GET', timeoutMs: DOWNLOAD_TIMEOUT_MS }); + const video = Buffer.from(await file.arrayBuffer()); + for (const warning of data.warnings ?? []) console.warn(`⚠️ [FFmpeg] ${warning}`); return { - success: true, - videoBase64: data.output_base64, + video, duration: data.duration, - ttsError: data.tts_error, - }; - - } catch (error) { - console.error('❌ FFmpeg Service error:', error); - return { - success: false, - error: error instanceof Error ? error.message : 'Unknown error', + ttsEngine: data.tts_engine, + ttsVoice: data.tts_voice, + warnings: data.warnings ?? [], }; + } finally { + // Le fichier n'est plus utile au service une fois récupéré + this.call(`/jobs/${data.job_id}`, { method: 'DELETE', timeoutMs: HEALTH_TIMEOUT_MS }) + .catch(() => { /* purgé plus tard par le service */ }); } } - /** - * Vérifie la santé de l'API FFmpeg - */ async healthCheck(): Promise { try { - const config = this.ensureConfigured(); - const response = await fetch(`${config.apiUrl}/health`, { - method: 'GET', - headers: { - 'X-API-Key': config.apiKey, - }, - signal: AbortSignal.timeout(HEALTH_TIMEOUT_MS), - }); - return response.ok; + await this.call('/health', { method: 'GET', timeoutMs: HEALTH_TIMEOUT_MS }); + return true; } catch { return false; } } - async previewTTS( + + /** Génère la voix seule (aperçu), avec le minutage de chaque mot. */ + async previewVoice( text: string, - ttsVoice?: string, - ttsEngine?: string, - geminiApiKey?: string - ): Promise<{ success: boolean; audioBase64?: string; error?: string }> { - const config = this.ensureConfigured(); - - try { - const response = await fetch(`${config.apiUrl}/preview-tts`, { - method: 'POST', - headers: { - 'Content-Type': 'application/json', - 'X-API-Key': config.apiKey, - }, - body: JSON.stringify({ - text, - tts_enabled: true, - tts_voice: ttsVoice, - tts_engine: ttsEngine, - gemini_api_key: geminiApiKey, - }), - signal: AbortSignal.timeout(TTS_TIMEOUT_MS), - }); - - if (!response.ok) { - const errorText = await response.text(); - return { success: false, error: `FFmpeg API error: ${response.status} - ${errorText}` }; - } - - const data = await response.json(); - - if (!data.success) { - return { success: false, error: data.detail }; - } - - return { success: true, audioBase64: data.audio_base64 }; - - } catch (error) { - console.error('❌ TTS Preview error:', error); - return { - success: false, - error: error instanceof Error ? error.message : 'Unknown error', - }; - } + options: { voice?: string; engine?: TtsEngine; style?: TtsStyle; geminiApiKey?: string } = {}, + ): Promise { + const response = await this.call('/preview-tts', { + method: 'POST', + headers: { 'Content-Type': 'application/json' }, + body: JSON.stringify({ + text, + tts_voice: options.voice, + tts_engine: options.engine, + tts_style: options.style, + gemini_api_key: options.geminiApiKey, + }), + timeoutMs: TTS_TIMEOUT_MS, + }); + const data = await response.json() as { + audio_base64: string; + duration: number; + words: TimedWord[]; + engine: string; + voice: string; + warnings?: string[]; + }; + return { + audio: Buffer.from(data.audio_base64, 'base64'), + duration: data.duration, + words: data.words ?? [], + engine: data.engine, + voice: data.voice, + warnings: data.warnings ?? [], + }; } } diff --git a/server/services/reels/imagesPipeline.ts b/server/services/reels/imagesPipeline.ts index da7ed58..b14e593 100644 --- a/server/services/reels/imagesPipeline.ts +++ b/server/services/reels/imagesPipeline.ts @@ -6,11 +6,10 @@ import fs from "fs"; import path from "path"; -import * as musicMetadata from "music-metadata"; import { bundle } from "@remotion/bundler"; import { renderMedia, selectComposition } from "@remotion/renderer"; import { imagesReelParamsSchema, type ImagesReelResult } from "@shared/reel"; -import { ffmpegService } from "../ffmpeg"; +import { ffmpegService, type TimedWord } from "../ffmpeg"; import { generateVideoThumbnail } from "../thumbnail"; import type { JobContext } from "./queue"; import { resolveGeminiApiKey, resolveLogoPath } from "./assets"; @@ -88,43 +87,31 @@ export function stripForTTS(text: string): string { .trim(); } -/** Nombre de syllabes d'un mot français (groupes de voyelles). */ -function countSyllablesFr(word: string): number { - const clean = word.replace(/[^a-zàâéèêëîïôùûüç]/gi, "").toLowerCase(); - if (!clean) return 1; - return Math.max(1, clean.match(/[aeiouyàâéèêëîïôùûü]+/gi)?.length ?? 1); +const EDGE_PUNCT = /^[.,!?;:…«»"'()\[\]]+|[.,!?;:…«»"'()\[\]]+$/g; + +export interface WordTiming { + word: string; + startFrame: number; + endFrame: number; } -const PUNCT_ONLY = /^[.,!?;:…\-—«»"''()\[\]]+$/; -const EDGE_PUNCT = /^[.,!?;:…«»"''()\[\]]+|[.,!?;:…«»"''()\[\]]+$/g; - /** - * Timings des mots prononcés (en images), répartis au prorata des syllabes. - * Estimation provisoire : remplacée par les vrais timings de la voix au lot - * « voix et sous-titres ». + * Convertit le minutage réel des mots (en secondes, fourni par la voix) en + * images. Chaque mot reste affiché jusqu'au début du suivant : pas de trou + * pendant les respirations. */ -export function computeWordTimings( - displayText: string, - audioDurationSeconds: number, - fps: number, - startFrame: number, -): Array<{ word: string; startFrame: number; endFrame: number }> { - const spokenWords = stripForTTS(displayText) - .split(/\s+/) - .filter((w) => w && !PUNCT_ONLY.test(w)); - if (spokenWords.length === 0) return []; - - const cleanWords = spokenWords.map((w) => w.replace(EDGE_PUNCT, "") || w); - const syllables = cleanWords.map(countSyllablesFr); - const totalSyllables = syllables.reduce((a, b) => a + b, 0); - const totalFrames = audioDurationSeconds * fps; - - let currentFrame = startFrame; - return cleanWords.map((word, i) => { - const wordStart = currentFrame; - currentFrame += Math.round((syllables[i] / totalSyllables) * totalFrames); - return { word, startFrame: wordStart, endFrame: currentFrame }; - }); +export function toWordTimings(words: TimedWord[], fps: number): WordTiming[] { + const timings = words + .map((w) => ({ word: w.text.replace(EDGE_PUNCT, "") || w.text, start: w.start, end: w.end })) + .filter((w) => w.word.trim()); + return timings.map((w, i) => ({ + word: w.word, + startFrame: Math.round(w.start * fps), + endFrame: Math.max( + Math.round(w.start * fps) + 1, + Math.round((i + 1 < timings.length ? timings[i + 1].start : w.end) * fps), + ), + })); } export async function runImagesReelJob({ job, progress }: JobContext): Promise { @@ -141,32 +128,27 @@ export async function runImagesReelJob({ job, progress }: JobContext): Promise | undefined; + let wordTimings: WordTiming[] | undefined; let audioDuration = 0; const ttsText = overlayText ? stripForTTS(overlayText) : ""; - if (overlayText && ttsText) { + if (params.ttsEnabled && overlayText && ttsText) { await progress(15, "voice"); - const geminiApiKey = await resolveGeminiApiKey(params.ttsEngine); - const tts = await ffmpegService.previewTTS(ttsText, params.ttsVoice, params.ttsEngine, geminiApiKey); - if (!tts.success || !tts.audioBase64) { - throw new Error(`La voix n'a pas pu être générée : ${tts.error ?? "réponse vide"}`); - } + const voice = await ffmpegService.previewVoice(ttsText, { + voice: params.ttsVoice, + engine: params.ttsEngine, + style: params.ttsStyle, + geminiApiKey: await resolveGeminiApiKey(params.ttsEngine), + }); + for (const warning of voice.warnings) console.warn(`⚠️ [Reels] ${warning}`); const audioFilename = `tts-${job.id}.mp3`; const audioPath = path.join(REMOTION_TEMP_DIR, audioFilename); - const audioBuffer = Buffer.from(tts.audioBase64, "base64"); - await fs.promises.writeFile(audioPath, audioBuffer); + await fs.promises.writeFile(audioPath, voice.audio); jobTempFiles.push(audioPath); audioUrl = localHttpUrl(`/uploads/temp/${audioFilename}`); - - try { - audioDuration = (await musicMetadata.parseFile(audioPath)).format.duration ?? 0; - } catch { - audioDuration = audioBuffer.length / 16000; // estimation à 128 kb/s - } - audioDuration = Math.max(audioDuration, ttsText.split(/\s+/).length * 0.35); - wordTimings = computeWordTimings(overlayText, audioDuration, FPS, 0); + audioDuration = voice.duration; + wordTimings = toWordTimings(voice.words, FPS); } // --- Durée : 25 à 30 s --- diff --git a/server/services/reels/queue.ts b/server/services/reels/queue.ts index f96a6cc..680536e 100644 --- a/server/services/reels/queue.ts +++ b/server/services/reels/queue.ts @@ -94,6 +94,7 @@ export async function startReelWorker(): Promise { console.error("❌ [ReelQueue] Récupération des jobs orphelins impossible :", error); } + console.log(`🎬 [ReelQueue] Worker démarré (traitements : ${Array.from(handlers.keys()).join(", ")})`); pollTimer = setInterval(kick, POLL_INTERVAL_MS); pollTimer.unref(); kick(); diff --git a/server/services/reels/videoPipeline.ts b/server/services/reels/videoPipeline.ts index 8562d1f..63445ce 100644 --- a/server/services/reels/videoPipeline.ts +++ b/server/services/reels/videoPipeline.ts @@ -30,14 +30,14 @@ export async function runVideoReelJob({ job, progress }: JobContext) { await progress(15, "render"); const startedAt = Date.now(); - const rendered = await ffmpegService.processReelFromUrl(resolveInternalUrl(media.originalUrl), { + const rendered = await ffmpegService.renderReel(resolveInternalUrl(media.originalUrl), { text: params.overlayText, musicUrl, ttsEnabled: params.ttsEnabled, ttsVoice: params.ttsVoice, ttsEngine: params.ttsEngine, + ttsStyle: params.ttsStyle, geminiApiKey, - wordDuration: params.wordDuration, fontSize: params.fontSize, musicVolume: params.musicVolume, drawText: params.drawText, @@ -46,18 +46,10 @@ export async function runVideoReelJob({ job, progress }: JobContext) { storeName: params.storeName, enableEndingEffect: params.enableEndingEffect, }); - console.log(`⏱️ [Reels] FFmpeg : ${((Date.now() - startedAt) / 1000).toFixed(1)} s`); - - if (!rendered.success || !rendered.videoBase64) { - throw new Error(rendered.error || "Erreur de traitement vidéo FFmpeg"); - } - if (rendered.ttsError) { - // Voix demandée mais absente : ne jamais publier un Reel muet sans le dire - throw new Error(`La voix n'a pas pu être générée : ${rendered.ttsError}`); - } + console.log(`⏱️ [Reels] Rendu : ${((Date.now() - startedAt) / 1000).toFixed(1)} s, vidéo de ${rendered.duration.toFixed(1)} s`); await progress(65, "store"); - const videoBuffer = Buffer.from(rendered.videoBase64, "base64"); + const videoBuffer = rendered.video; const processedMedia = await storeRenderedVideo(job.userId, videoBuffer, `reel-${Date.now()}.mp4`); await storage.updatePostMedia(postId, [processedMedia.id]); diff --git a/server/services/ttsSync.ts b/server/services/ttsSync.ts index 8113d28..33a85a5 100644 --- a/server/services/ttsSync.ts +++ b/server/services/ttsSync.ts @@ -1,5 +1,15 @@ -import { ffmpegService } from './ffmpeg'; -import * as musicMetadata from 'music-metadata'; +/** + * Estimation de la durée de lecture d'un texte, affichée pendant la saisie. + * + * Auparavant, chaque pause de frappe générait une voix complète pour la + * mesurer (un appel Gemini facturé à chaque fois). Le minutage réel des mots + * vient désormais de la voix au moment du rendu ; ici, une estimation suffit. + */ + +/** Débit moyen d'une voix de synthèse française, en mots par seconde. */ +const WORDS_PER_SECOND = 2.6; +/** Au-delà, la voix dépasse la durée confortable d'un Reel. */ +const MAX_COMFORTABLE_SECONDS = 45; export interface SyncTiming { wordDuration: number; @@ -10,67 +20,30 @@ export interface SyncTiming { warnings: string[]; } -export class TtsSyncService { - /** - * Calcule le word_duration optimal pour synchroniser l'affichage du texte - * avec la durée réelle de la voix TTS générée. - */ - async calculateSyncTiming( - text: string, - voice: string, - ttsEngine?: string, - geminiApiKey?: string - ): Promise { - const cleanText = this.cleanText(text); +export function estimateVoiceTiming(text: string): SyncTiming { + const words = text + .replace(/#[\wÀ-ÿ]+/g, '') + .replace(/https?:\/\/\S+/g, '') + .split(/\s+/) + .filter((w) => /[A-Za-z0-9À-ÿ]/.test(w)); + const pauses = (text.match(/[.!?;:]/g) ?? []).length; + const punctuationPause = 0.35; + const audioDuration = words.length / WORDS_PER_SECOND + pauses * punctuationPause; - // 1. Générer le TTS preview (avec la même voix que le rendu) et mesurer sa durée exacte - const ttsResult = await ffmpegService.previewTTS(cleanText, voice, ttsEngine, geminiApiKey); - if (!ttsResult.success || !ttsResult.audioBase64) { - throw new Error('TTS preview failed: ' + (ttsResult.error || 'unknown')); - } - - const audioBuffer = Buffer.from(ttsResult.audioBase64, 'base64'); - const metadata = await musicMetadata.parseBuffer(audioBuffer, 'audio/mpeg'); - const audioDuration = metadata.format.duration || 0; - - // 2. Analyser le texte (compte les mots réellement lus par la voix) - const wordCount = this.calculateWordCount(cleanText); - - // 3. Calculer le word_duration - const wordDuration = wordCount > 0 ? audioDuration / wordCount : 0.6; - - // 4. Validation - const warnings: string[] = []; - const isHealthy = wordDuration >= 0.25 && wordDuration <= 1.5; - if (wordDuration < 0.25) { - warnings.push('Texte trop long : les mots défileront très vite. Envisagez de raccourcir.'); - } - if (wordDuration > 1.2) { - warnings.push('Texte très court : affichage lent.'); - } - - return { - wordDuration, - audioDuration, - wordCount, - punctuationPause: 0, - isHealthy, - warnings, - }; + const warnings: string[] = []; + if (audioDuration > MAX_COMFORTABLE_SECONDS) { + warnings.push(`Texte long : environ ${Math.round(audioDuration)} s de voix. Visez moins de ${MAX_COMFORTABLE_SECONDS} s.`); + } + if (words.length > 0 && words.length < 4) { + warnings.push('Texte très court : la voix ne durera que quelques secondes.'); } - private cleanText(text: string): string { - return text - .replace(/#\w+/g, '') - .replace(/https?:\/\/\S+/g, '') - .replace(/[\uD83C-\uD83E][\uDC00-\uDFFF]|[☀-⛿✀-➿]/g, '') - .trim(); - } - - private calculateWordCount(text: string): number { - const tokens = text.split(/\s+/).filter(w => w.length > 0); - return tokens.filter(w => /[a-zA-Z0-9À-ſ]/.test(w)).length; - } + return { + wordDuration: words.length ? audioDuration / words.length : 0, + audioDuration, + wordCount: words.length, + punctuationPause, + isHealthy: warnings.length === 0, + warnings, + }; } - -export const ttsSyncService = new TtsSyncService(); diff --git a/shared/reel.ts b/shared/reel.ts index 6796313..d972964 100644 --- a/shared/reel.ts +++ b/shared/reel.ts @@ -5,6 +5,7 @@ */ import { z } from "zod"; +import { TTS_ENGINES, TTS_STYLES } from "./voices"; const optionalText = z .string() @@ -24,9 +25,11 @@ export const videoReelParamsSchema = z.object({ description: optionalText, ttsEnabled: z.boolean().default(false), ttsVoice: optionalText, - ttsEngine: z.enum(["gemini", "edge"]).optional(), + ttsEngine: z.enum(TTS_ENGINES).optional(), + ttsStyle: z.enum(TTS_STYLES).optional(), scheduledFor: optionalText, - wordDuration: z.number().positive().max(5).default(0.6), + // Ancien réglage, ignoré : le minutage vient désormais de la voix elle-même + wordDuration: z.number().optional(), fontSize: z.number().int().min(16).max(200).default(64), musicVolume: z.number().min(0).max(2).default(0.25), drawText: z.boolean().default(true), @@ -45,13 +48,23 @@ export const imagesReelParamsSchema = z.object({ overlayText: optionalText, musicUrl: optionalText, musicVolume: z.number().min(0).max(2).default(0.3), - ttsEngine: z.enum(["gemini", "edge"]).optional(), + ttsEnabled: z.boolean().default(true), + ttsEngine: z.enum(TTS_ENGINES).optional(), ttsVoice: optionalText, + ttsStyle: z.enum(TTS_STYLES).optional(), storeName: z.string().optional(), // Fichiers temporaires à supprimer une fois le rendu terminé tempFiles: z.array(z.string()).default([]), }); +/** Aperçu de la voix. */ +export const ttsPreviewSchema = z.object({ + text: z.string({ required_error: "Texte requis" }).trim().min(1, "Texte requis").max(2000), + ttsVoice: optionalText, + ttsEngine: z.enum(TTS_ENGINES).optional(), + ttsStyle: z.enum(TTS_STYLES).optional(), +}); + export type ImagesReelParams = z.infer; export type ReelJobKind = "video" | "images"; diff --git a/shared/voices.ts b/shared/voices.ts new file mode 100644 index 0000000..c43f9d3 --- /dev/null +++ b/shared/voices.ts @@ -0,0 +1,76 @@ +/** + * Voix et styles de lecture proposés pour les Reels. + * Même catalogue que le service Python (ffmpeg-service/app/voices.py). + */ + +export const TTS_ENGINES = ["gemini", "edge"] as const; +export type TtsEngine = (typeof TTS_ENGINES)[number]; + +export const TTS_STYLES = ["neutral", "dynamic", "warm", "calm", "promo"] as const; +export type TtsStyle = (typeof TTS_STYLES)[number]; + +export interface VoiceOption { + id: string; + label: string; + gender: "female" | "male"; +} + +export const TTS_STYLE_OPTIONS: { id: TtsStyle; label: string; description: string }[] = [ + { id: "dynamic", label: "Dynamique", description: "Enthousiaste, rythme entraînant" }, + { id: "warm", label: "Chaleureux", description: "Souriant et proche" }, + { id: "promo", label: "Promo", description: "Annonce énergique des offres" }, + { id: "calm", label: "Calme", description: "Posé et rassurant" }, + { id: "neutral", label: "Neutre", description: "Lecture sans consigne" }, +]; + +export const GEMINI_VOICES: VoiceOption[] = [ + { id: "Kore", label: "Kore — ferme", gender: "female" }, + { id: "Aoede", label: "Aoede — légère", gender: "female" }, + { id: "Leda", label: "Leda — jeune", gender: "female" }, + { id: "Zephyr", label: "Zephyr — lumineuse", gender: "female" }, + { id: "Callirrhoe", label: "Callirrhoe — décontractée", gender: "female" }, + { id: "Autonoe", label: "Autonoe — lumineuse", gender: "female" }, + { id: "Despina", label: "Despina — douce", gender: "female" }, + { id: "Erinome", label: "Erinome — claire", gender: "female" }, + { id: "Laomedeia", label: "Laomedeia — enjouée", gender: "female" }, + { id: "Achernar", label: "Achernar — tendre", gender: "female" }, + { id: "Gacrux", label: "Gacrux — mûre", gender: "female" }, + { id: "Pulcherrima", label: "Pulcherrima — affirmée", gender: "female" }, + { id: "Vindemiatrix", label: "Vindemiatrix — délicate", gender: "female" }, + { id: "Sulafat", label: "Sulafat — chaleureuse", gender: "female" }, + { id: "Charon", label: "Charon — informative", gender: "male" }, + { id: "Puck", label: "Puck — enjouée", gender: "male" }, + { id: "Fenrir", label: "Fenrir — enthousiaste", gender: "male" }, + { id: "Orus", label: "Orus — ferme", gender: "male" }, + { id: "Enceladus", label: "Enceladus — soufflée", gender: "male" }, + { id: "Iapetus", label: "Iapetus — claire", gender: "male" }, + { id: "Umbriel", label: "Umbriel — décontractée", gender: "male" }, + { id: "Algieba", label: "Algieba — veloutée", gender: "male" }, + { id: "Algenib", label: "Algenib — rocailleuse", gender: "male" }, + { id: "Rasalgethi", label: "Rasalgethi — informative", gender: "male" }, + { id: "Alnilam", label: "Alnilam — ferme", gender: "male" }, + { id: "Schedar", label: "Schedar — posée", gender: "male" }, + { id: "Achird", label: "Achird — amicale", gender: "male" }, + { id: "Zubenelgenubi", label: "Zubenelgenubi — naturelle", gender: "male" }, + { id: "Sadachbia", label: "Sadachbia — vive", gender: "male" }, + { id: "Sadaltager", label: "Sadaltager — experte", gender: "male" }, +]; + +export const EDGE_VOICES: VoiceOption[] = [ + { id: "fr-FR-VivienneMultilingualNeural", label: "Vivienne", gender: "female" }, + { id: "fr-FR-DeniseNeural", label: "Denise", gender: "female" }, + { id: "fr-FR-EloiseNeural", label: "Eloise", gender: "female" }, + { id: "fr-FR-RemyMultilingualNeural", label: "Rémy", gender: "male" }, + { id: "fr-FR-HenriNeural", label: "Henri", gender: "male" }, +]; + +export const DEFAULT_VOICE: Record = { + gemini: "Kore", + edge: "fr-FR-VivienneMultilingualNeural", +}; + +export const DEFAULT_TTS_STYLE: TtsStyle = "dynamic"; + +export function voicesFor(engine: TtsEngine): VoiceOption[] { + return engine === "gemini" ? GEMINI_VOICES : EDGE_VOICES; +}