mirror of
https://github.com/R0m1k3/Socialflow.git
synced 2026-10-11 17:26:45 +02:00
feat(reels): voix Gemini complète, sous-titres mot à mot et rendu de qualité
Service ffmpeg-api réécrit (ffmpeg-service/app/) : - Appels FFmpeg asynchrones : le service ne se fige plus pendant un rendu - Vidéo récupérée par téléchargement (GET /files/…) au lieu de base64 en JSON - Vraies erreurs HTTP ; échec explicite si la voix demandée est impossible - 30 voix Gemini + ton de lecture (dynamique, chaleureux, promo, calme), clé en en-tête, modèle configurable avec repli - Edge TTS 7 (minutage des mots restauré), secours en voix françaises - Voix traitée : filtre, compression, niveau constant ; musique bouclée et baissée automatiquement sous la voix ; mix final à -14 LUFS - Sous-titres calés mot à mot (Whisper pour Gemini et la voix d'origine, à la place de ffsubsync), style Montserrat, placés hors des boutons Reels - Vidéo : plus de retouche luminosité forcée, scaling lanczos, HDR iPhone converti, BT.709, AAC 48 kHz 192k ; la vidéo s'allonge si la voix dépasse - Grand logo de fin affiché après la voix ; FFmpeg 7.0.2 épinglé, polices et modèle Whisper intégrés à l'image ; tests pytest et ruff Application : - Sélecteur de voix partagé (4 pages) : moteur, 30 voix, ton, écoute - Reel images : minutage réel des mots, interrupteur voix respecté - sync-info ne génère plus de voix à chaque frappe (estimation locale) - Stabilisation désactivée par défaut, route /reels/preview inutilisée retirée - Log « [ReelQueue] Worker démarré » pour vérifier la version déployée Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_018Ze4bs7tpF1KGWUk6ZZSZ4
This commit is contained in:
47 files changed
+2206
-2285
No files matched your search
@@ -105,3 +105,22 @@ LEGAL_CONTACT_EMAIL=
|
||||
# Clé d'accès pour l'API externe /api/v1/*
|
||||
# Générer avec: openssl rand -hex 32
|
||||
EXTERNAL_API_KEY=your-secure-random-key-here
|
||||
|
||||
# ==========================================
|
||||
# REELS : VOIX ET RENDU VIDÉO (service ffmpeg-api)
|
||||
# ==========================================
|
||||
# Clé partagée entre l'application et le service FFmpeg (CHANGEZ CETTE VALEUR).
|
||||
FFMPEG_API_KEY=change-me-ffmpeg-key
|
||||
#
|
||||
# Clé Google AI Studio pour les voix Gemini (sinon : voix Edge, gratuite).
|
||||
# Peut aussi être saisie dans l'application, ce qui est prioritaire.
|
||||
GEMINI_API_KEY=
|
||||
#
|
||||
# Modèle Gemini TTS (défaut : gemini-2.5-flash-preview-tts ; repli automatique
|
||||
# sur celui-ci si le modèle choisi n'existe pas).
|
||||
# GEMINI_TTS_MODEL=
|
||||
#
|
||||
# Modèle Whisper utilisé pour caler les sous-titres mot à mot sur la voix
|
||||
# (tiny, base, small). Choisi à la construction de l'image : changer la valeur
|
||||
# demande de reconstruire ffmpeg-api.
|
||||
# WHISPER_MODEL=base
|
||||
@@ -174,6 +174,20 @@ La stack se construit depuis les sources : `app` et `ffmpeg-api` ont une section
|
||||
`socialflow-ffmpeg-api:latest`) : il ne dépend donc plus du nom donné à la
|
||||
stack dans Portainer.
|
||||
|
||||
## 🎙️ Service ffmpeg-api (voix et rendu des Reels)
|
||||
|
||||
- **Premier build plus long** : l'image embarque FFmpeg 7.0.2 (version
|
||||
épinglée), les polices des sous-titres et le modèle Whisper qui cale les
|
||||
sous-titres mot à mot sur la voix (~150 Mo avec `WHISPER_MODEL=base`).
|
||||
- **Voix Gemini** : renseignez `GEMINI_API_KEY` (ou la clé dans l'application).
|
||||
Sans clé, la voix Edge gratuite est utilisée et un avertissement apparaît
|
||||
dans les logs.
|
||||
- **Vérifier la version déployée** : au démarrage, les logs de `socialflow-app`
|
||||
affichent `[ReelQueue] Worker démarré`, et `socialflow-ffmpeg` répond
|
||||
`{"status":"ok","version":2}` sur `/health`.
|
||||
- **Tests du service** : `pip install -r requirements-dev.txt`, puis `pytest`
|
||||
et `ruff check .` dans `ffmpeg-service/`.
|
||||
|
||||
## 🔒 Sécurité en production
|
||||
|
||||
1. **Variables d'environnement** : Ne commitez JAMAIS le fichier `.env`
|
||||
|
||||
@@ -0,0 +1,185 @@
|
||||
import { useEffect, useRef, useState } from "react";
|
||||
import { Loader2, Play, Square } from "lucide-react";
|
||||
import { Button } from "@/components/ui/button";
|
||||
import { Label } from "@/components/ui/label";
|
||||
import {
|
||||
Select,
|
||||
SelectContent,
|
||||
SelectGroup,
|
||||
SelectItem,
|
||||
SelectLabel,
|
||||
SelectTrigger,
|
||||
SelectValue,
|
||||
} from "@/components/ui/select";
|
||||
import { useToast } from "@/hooks/use-toast";
|
||||
import { apiRequest, getErrorMessage } from "@/lib/queryClient";
|
||||
import {
|
||||
DEFAULT_VOICE,
|
||||
TTS_STYLE_OPTIONS,
|
||||
voicesFor,
|
||||
type TtsEngine,
|
||||
type TtsStyle,
|
||||
} from "@shared/voices";
|
||||
|
||||
export interface VoiceSettings {
|
||||
engine: TtsEngine;
|
||||
voice: string;
|
||||
style: TtsStyle;
|
||||
}
|
||||
|
||||
interface VoicePickerProps {
|
||||
value: VoiceSettings;
|
||||
onChange: (value: VoiceSettings) => void;
|
||||
/** Texte lu par le bouton « Tester la voix ». */
|
||||
sampleText?: string;
|
||||
compact?: boolean;
|
||||
}
|
||||
|
||||
const FALLBACK_SAMPLE = "Découvrez nos nouveautés en magasin, on vous attend !";
|
||||
|
||||
/**
|
||||
* Choix du moteur, de la voix et du ton, avec écoute de l'aperçu.
|
||||
* Partagé par les pages Reel (vidéo et images, bureau et mobile).
|
||||
*/
|
||||
export function VoicePicker({ value, onChange, sampleText, compact = false }: VoicePickerProps) {
|
||||
const { toast } = useToast();
|
||||
const [loading, setLoading] = useState(false);
|
||||
const [playing, setPlaying] = useState(false);
|
||||
const audioRef = useRef<HTMLAudioElement | null>(null);
|
||||
|
||||
useEffect(() => () => audioRef.current?.pause(), []);
|
||||
|
||||
const voices = voicesFor(value.engine);
|
||||
const groups = [
|
||||
{ label: "Voix féminines", items: voices.filter((v) => v.gender === "female") },
|
||||
{ label: "Voix masculines", items: voices.filter((v) => v.gender === "male") },
|
||||
];
|
||||
|
||||
const setEngine = (engine: TtsEngine) => {
|
||||
if (engine !== value.engine) onChange({ ...value, engine, voice: DEFAULT_VOICE[engine] });
|
||||
};
|
||||
|
||||
const stop = () => {
|
||||
audioRef.current?.pause();
|
||||
setPlaying(false);
|
||||
};
|
||||
|
||||
const preview = async () => {
|
||||
if (playing) return stop();
|
||||
setLoading(true);
|
||||
try {
|
||||
const response = await apiRequest("POST", "/api/reels/tts-preview", {
|
||||
text: sampleText?.trim() || FALLBACK_SAMPLE,
|
||||
ttsEngine: value.engine,
|
||||
ttsVoice: value.voice,
|
||||
ttsStyle: value.style,
|
||||
});
|
||||
const data = await response.json();
|
||||
for (const warning of data.warnings ?? []) {
|
||||
toast({ title: "Voix de secours utilisée", description: warning });
|
||||
}
|
||||
const audio = new Audio(`data:audio/mpeg;base64,${data.audioBase64}`);
|
||||
audioRef.current?.pause();
|
||||
audioRef.current = audio;
|
||||
audio.onended = () => setPlaying(false);
|
||||
await audio.play();
|
||||
setPlaying(true);
|
||||
} catch (error) {
|
||||
toast({
|
||||
title: "Impossible de tester la voix",
|
||||
description: getErrorMessage(error, "La voix n'a pas pu être générée."),
|
||||
variant: "destructive",
|
||||
});
|
||||
} finally {
|
||||
setLoading(false);
|
||||
}
|
||||
};
|
||||
|
||||
const labelClass = compact ? "text-xs font-medium" : "text-sm font-medium";
|
||||
|
||||
return (
|
||||
<div className="space-y-3">
|
||||
<div className="flex items-center gap-3">
|
||||
<Label className={`${labelClass} w-14 shrink-0`}>Moteur</Label>
|
||||
<div className="flex gap-2">
|
||||
<Button
|
||||
type="button"
|
||||
size="sm"
|
||||
className={compact ? "h-7 text-xs" : undefined}
|
||||
variant={value.engine === "gemini" ? "default" : "outline"}
|
||||
onClick={() => setEngine("gemini")}
|
||||
>
|
||||
Gemini (naturelle)
|
||||
</Button>
|
||||
<Button
|
||||
type="button"
|
||||
size="sm"
|
||||
className={compact ? "h-7 text-xs" : undefined}
|
||||
variant={value.engine === "edge" ? "default" : "outline"}
|
||||
onClick={() => setEngine("edge")}
|
||||
>
|
||||
Edge (gratuite)
|
||||
</Button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div className="flex items-center gap-3">
|
||||
<Label className={`${labelClass} w-14 shrink-0`}>Voix</Label>
|
||||
<Select value={value.voice} onValueChange={(voice) => onChange({ ...value, voice })}>
|
||||
<SelectTrigger className={compact ? "h-8 text-xs flex-1" : "flex-1"}>
|
||||
<SelectValue placeholder="Choisir une voix" />
|
||||
</SelectTrigger>
|
||||
<SelectContent className="max-h-72">
|
||||
{groups.map((group) => (
|
||||
<SelectGroup key={group.label}>
|
||||
<SelectLabel>{group.label}</SelectLabel>
|
||||
{group.items.map((voice) => (
|
||||
<SelectItem key={voice.id} value={voice.id}>
|
||||
{voice.label}
|
||||
</SelectItem>
|
||||
))}
|
||||
</SelectGroup>
|
||||
))}
|
||||
</SelectContent>
|
||||
</Select>
|
||||
</div>
|
||||
|
||||
<div className="flex items-center gap-3">
|
||||
<Label className={`${labelClass} w-14 shrink-0`}>Ton</Label>
|
||||
<Select value={value.style} onValueChange={(style) => onChange({ ...value, style: style as TtsStyle })}>
|
||||
<SelectTrigger className={compact ? "h-8 text-xs flex-1" : "flex-1"}>
|
||||
<SelectValue />
|
||||
</SelectTrigger>
|
||||
<SelectContent>
|
||||
{TTS_STYLE_OPTIONS.map((style) => (
|
||||
<SelectItem key={style.id} value={style.id}>
|
||||
{style.label} <span className="text-muted-foreground">— {style.description}</span>
|
||||
</SelectItem>
|
||||
))}
|
||||
</SelectContent>
|
||||
</Select>
|
||||
</div>
|
||||
|
||||
<Button
|
||||
type="button"
|
||||
size="sm"
|
||||
variant="secondary"
|
||||
className={compact ? "w-full h-8 text-xs" : "w-full"}
|
||||
disabled={loading}
|
||||
onClick={(e) => {
|
||||
e.stopPropagation();
|
||||
preview();
|
||||
}}
|
||||
>
|
||||
{loading ? (
|
||||
<Loader2 className="w-3 h-3 mr-2 animate-spin" />
|
||||
) : playing ? (
|
||||
<Square className="w-3 h-3 mr-2" />
|
||||
) : (
|
||||
<Play className="w-3 h-3 mr-2" />
|
||||
)}
|
||||
{loading ? "Génération de la voix…" : playing ? "Arrêter" : "Tester la voix"}
|
||||
</Button>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
@@ -16,6 +16,8 @@ import { Switch } from "@/components/ui/switch";
|
||||
import { Label } from "@/components/ui/label";
|
||||
import { Slider } from "@/components/ui/slider";
|
||||
import { useToast } from "@/hooks/use-toast";
|
||||
import { VoicePicker, type VoiceSettings } from "@/components/reels/voice-picker";
|
||||
import { DEFAULT_TTS_STYLE, DEFAULT_VOICE } from "@shared/voices";
|
||||
import { apiRequest, queryClient, handleUnauthorized } from "@/lib/queryClient";
|
||||
import type { SocialPage, Media } from "@shared/schema";
|
||||
import { SiFacebook, SiTiktok } from "react-icons/si";
|
||||
@@ -59,27 +61,17 @@ export default function MobileNewReel() {
|
||||
|
||||
// État TTS
|
||||
const [ttsEnabled, setTtsEnabled] = useState(false);
|
||||
const [ttsEngine, setTtsEngine] = useState<'edge' | 'gemini'>('edge');
|
||||
const [ttsVoice, setTtsVoice] = useState('fr-FR-VivienneMultilingualNeural');
|
||||
|
||||
// Gemini native TTS voices (Charon = homme, Kore = femme)
|
||||
const geminiVoices = [
|
||||
{ label: 'Charon - Voix Homme', value: 'fr-FR-Standard-B' },
|
||||
{ label: 'Kore - Voix Femme', value: 'fr-FR-Standard-A' },
|
||||
];
|
||||
|
||||
// French Edge TTS voices
|
||||
const edgeVoices = [
|
||||
{ label: 'Vivienne (Femme)', value: 'fr-FR-VivienneMultilingualNeural' },
|
||||
{ label: 'Henri (Homme)', value: 'fr-FR-HenriNeural' },
|
||||
{ label: 'Denise (Femme)', value: 'fr-FR-DeniseNeural' },
|
||||
{ label: 'Rémy (Homme)', value: 'fr-FR-RemyMultilingualNeural' },
|
||||
{ label: 'Jenny (Anglaise, Femme)', value: 'en-US-JennyNeural' },
|
||||
];
|
||||
const [voiceSettings, setVoiceSettings] = useState<VoiceSettings>({
|
||||
engine: 'gemini',
|
||||
voice: DEFAULT_VOICE.gemini,
|
||||
style: DEFAULT_TTS_STYLE,
|
||||
});
|
||||
const { engine: ttsEngine, voice: ttsVoice, style: ttsStyle } = voiceSettings;
|
||||
|
||||
// État audio preview
|
||||
const [isPlaying, setIsPlaying] = useState<string | null>(null);
|
||||
const [stabilize, setStabilize] = useState(true); // Activé par défaut pour les Reels
|
||||
// Désactivée par défaut : double le temps de rendu, utile seulement pour une vidéo tremblée
|
||||
const [stabilize, setStabilize] = useState(false);
|
||||
const [enableEndingEffect, setEnableEndingEffect] = useState(true);
|
||||
|
||||
// TTS Sync state
|
||||
@@ -271,6 +263,7 @@ export default function MobileNewReel() {
|
||||
ttsEnabled,
|
||||
ttsEngine,
|
||||
ttsVoice,
|
||||
ttsStyle,
|
||||
enableEndingEffect,
|
||||
});
|
||||
};
|
||||
@@ -492,82 +485,13 @@ export default function MobileNewReel() {
|
||||
<span>TTS — voix activée</span>
|
||||
</div>
|
||||
|
||||
{/* TTS Engine Selector */}
|
||||
<div className="flex items-center gap-2 mt-2">
|
||||
<Label className="text-xs font-medium">Moteur:</Label>
|
||||
<div className="flex gap-1">
|
||||
<Button
|
||||
size="sm"
|
||||
variant={ttsEngine === 'edge' ? 'default' : 'outline'}
|
||||
className="h-7 text-xs px-2"
|
||||
onClick={() => {
|
||||
setTtsEngine('edge');
|
||||
setTtsVoice('fr-FR-VivienneMultilingualNeural');
|
||||
}}
|
||||
>
|
||||
Edge
|
||||
</Button>
|
||||
<Button
|
||||
size="sm"
|
||||
variant={ttsEngine === 'gemini' ? 'default' : 'outline'}
|
||||
className="h-7 text-xs px-2"
|
||||
onClick={() => {
|
||||
setTtsEngine('gemini');
|
||||
setTtsVoice('fr-FR-Standard-B');
|
||||
}}
|
||||
>
|
||||
Gemini
|
||||
</Button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{/* Voice Selector */}
|
||||
<div className="flex items-center gap-2 mt-2">
|
||||
<Label className="text-xs font-medium shrink-0">Voix:</Label>
|
||||
<Select value={ttsVoice} onValueChange={setTtsVoice}>
|
||||
<SelectTrigger className="h-7 text-xs flex-1">
|
||||
<SelectValue />
|
||||
</SelectTrigger>
|
||||
<SelectContent>
|
||||
{ttsEngine === 'edge' ? (
|
||||
edgeVoices.map(v => (
|
||||
<SelectItem key={v.value} value={v.value}>{v.label}</SelectItem>
|
||||
))
|
||||
) : (
|
||||
geminiVoices.map(v => (
|
||||
<SelectItem key={v.value} value={v.value}>{v.label}</SelectItem>
|
||||
))
|
||||
)}
|
||||
</SelectContent>
|
||||
</Select>
|
||||
</div>
|
||||
|
||||
<div className="mt-2">
|
||||
<Button
|
||||
size="sm"
|
||||
variant="secondary"
|
||||
className="w-full h-8 text-xs"
|
||||
onClick={async (e) => {
|
||||
e.stopPropagation();
|
||||
const textToTest = overlayText || "Ceci est un test de voix.";
|
||||
try {
|
||||
const response = await apiRequest('POST', '/api/reels/tts-preview', {
|
||||
text: textToTest,
|
||||
ttsEngine,
|
||||
ttsVoice,
|
||||
});
|
||||
const data = await response.json();
|
||||
if (data.success && data.audioBase64) {
|
||||
const audio = new Audio(`data:audio/mp3;base64,${data.audioBase64}`);
|
||||
audio.play();
|
||||
}
|
||||
} catch (err) {
|
||||
toast({ title: "Erreur", description: "Impossible de lire la voix", variant: "destructive" });
|
||||
}
|
||||
}}
|
||||
>
|
||||
<Play className="w-3 h-3 mr-1" /> Tester la voix
|
||||
</Button>
|
||||
<VoicePicker
|
||||
value={voiceSettings}
|
||||
onChange={setVoiceSettings}
|
||||
sampleText={overlayText}
|
||||
compact
|
||||
/>
|
||||
</div>
|
||||
|
||||
{syncInfo && (
|
||||
|
||||
@@ -1,5 +1,7 @@
|
||||
import { useState, useRef } from "react";
|
||||
import { useToast } from "@/hooks/use-toast";
|
||||
import { VoicePicker, type VoiceSettings } from "@/components/reels/voice-picker";
|
||||
import { DEFAULT_TTS_STYLE, DEFAULT_VOICE } from "@shared/voices";
|
||||
import { Card, CardContent, CardHeader, CardTitle } from "@/components/ui/card";
|
||||
import { Button } from "@/components/ui/button";
|
||||
import { Textarea } from "@/components/ui/textarea";
|
||||
@@ -33,23 +35,11 @@ export default function MobileRemotionVideoPage() {
|
||||
const [productInfo, setProductInfo] = useState("");
|
||||
const [generatedVariants, setGeneratedVariants] = useState<any[]>([]);
|
||||
const [ttsEnabled, setTtsEnabled] = useState(true);
|
||||
const [ttsEngine, setTtsEngine] = useState<'edge' | 'gemini'>('edge');
|
||||
const [ttsVoice, setTtsVoice] = useState('fr-FR-VivienneMultilingualNeural');
|
||||
|
||||
// Gemini native TTS voices (Charon = homme, Kore = femme)
|
||||
const geminiVoices = [
|
||||
{ label: 'Charon - Voix Homme', value: 'fr-FR-Standard-B' },
|
||||
{ label: 'Kore - Voix Femme', value: 'fr-FR-Standard-A' },
|
||||
];
|
||||
|
||||
// French Edge TTS voices
|
||||
const edgeVoices = [
|
||||
{ label: 'Vivienne (Femme)', value: 'fr-FR-VivienneMultilingualNeural' },
|
||||
{ label: 'Henri (Homme)', value: 'fr-FR-HenriNeural' },
|
||||
{ label: 'Denise (Femme)', value: 'fr-FR-DeniseNeural' },
|
||||
{ label: 'Rémy (Homme)', value: 'fr-FR-RemyMultilingualNeural' },
|
||||
{ label: 'Jenny (Anglaise, Femme)', value: 'en-US-JennyNeural' },
|
||||
];
|
||||
const [voiceSettings, setVoiceSettings] = useState<VoiceSettings>({
|
||||
engine: 'gemini',
|
||||
voice: DEFAULT_VOICE.gemini,
|
||||
style: DEFAULT_TTS_STYLE,
|
||||
});
|
||||
const [musicFile, setMusicFile] = useState<File | null>(null);
|
||||
const [selectedTrack, setSelectedTrack] = useState<AudioTrack | null>(null);
|
||||
const [musicVolume, setMusicVolume] = useState(0.3);
|
||||
@@ -143,16 +133,6 @@ export default function MobileRemotionVideoPage() {
|
||||
|
||||
const totalSelected = images.length + selectedLibraryImages.length;
|
||||
|
||||
const handleTtsPreview = async () => {
|
||||
const ttsText = stripForTTS(overlayText);
|
||||
if (!ttsText) return;
|
||||
try {
|
||||
const r = await apiRequest('POST', '/api/reels/tts-preview', { text: ttsText, ttsEngine, ttsVoice });
|
||||
const data = await r.json();
|
||||
if (data.success && data.audioBase64) new window.Audio(`data:audio/mp3;base64,${data.audioBase64}`).play();
|
||||
} catch { toast({ title: "Erreur prévisualisation voix", variant: "destructive" }); }
|
||||
};
|
||||
|
||||
const togglePlayPreview = (track: AudioTrack) => {
|
||||
if (!audioRef.current) return;
|
||||
if (isPlaying === track.id) { audioRef.current.pause(); setIsPlaying(null); }
|
||||
@@ -168,8 +148,10 @@ export default function MobileRemotionVideoPage() {
|
||||
images.forEach(img => formData.append("images", img));
|
||||
selectedLibraryImages.forEach(m => formData.append("existingImageUrls", m.originalUrl));
|
||||
if (overlayText) formData.append("overlayText", overlayText);
|
||||
formData.append("ttsEngine", ttsEngine);
|
||||
formData.append("ttsVoice", ttsVoice);
|
||||
formData.append("ttsEnabled", String(ttsEnabled));
|
||||
formData.append("ttsEngine", voiceSettings.engine);
|
||||
formData.append("ttsVoice", voiceSettings.voice);
|
||||
formData.append("ttsStyle", voiceSettings.style);
|
||||
if (selectedPageIds[0]) formData.append("selectedPageId", selectedPageIds[0]);
|
||||
if (musicFile) { formData.append("music", musicFile); formData.append("musicVolume", String(musicVolume)); }
|
||||
else if (selectedTrack) { formData.append("musicTrackUrl", selectedTrack.url); formData.append("musicVolume", String(musicVolume)); }
|
||||
@@ -354,41 +336,7 @@ export default function MobileRemotionVideoPage() {
|
||||
<div className="space-y-2">
|
||||
<p className="text-xs text-muted-foreground">TTS — voix activée</p>
|
||||
|
||||
{/* TTS Engine Selector */}
|
||||
<div className="flex items-center gap-2">
|
||||
<Label className="text-xs font-medium">Moteur:</Label>
|
||||
<div className="flex gap-1">
|
||||
<Button size="sm" variant={ttsEngine === 'edge' ? 'default' : 'outline'} className="h-7 text-xs px-2"
|
||||
onClick={() => { setTtsEngine('edge'); setTtsVoice('fr-FR-VivienneMultilingualNeural'); }}>
|
||||
Edge
|
||||
</Button>
|
||||
<Button size="sm" variant={ttsEngine === 'gemini' ? 'default' : 'outline'} className="h-7 text-xs px-2"
|
||||
onClick={() => { setTtsEngine('gemini'); setTtsVoice('fr-FR-Standard-B'); }}>
|
||||
Gemini
|
||||
</Button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{/* Voice Selector */}
|
||||
<div className="flex items-center gap-2">
|
||||
<Label className="text-xs font-medium shrink-0">Voix:</Label>
|
||||
<Select value={ttsVoice} onValueChange={setTtsVoice}>
|
||||
<SelectTrigger className="h-7 text-xs flex-1">
|
||||
<SelectValue />
|
||||
</SelectTrigger>
|
||||
<SelectContent>
|
||||
{ttsEngine === 'edge' ? (
|
||||
edgeVoices.map(v => <SelectItem key={v.value} value={v.value}>{v.label}</SelectItem>)
|
||||
) : (
|
||||
geminiVoices.map(v => <SelectItem key={v.value} value={v.value}>{v.label}</SelectItem>)
|
||||
)}
|
||||
</SelectContent>
|
||||
</Select>
|
||||
</div>
|
||||
|
||||
<Button size="sm" variant="outline" className="w-full" onClick={handleTtsPreview} disabled={!overlayText}>
|
||||
<Volume2 className="mr-2 w-3 h-3" /> Écouter la voix
|
||||
</Button>
|
||||
<VoicePicker value={voiceSettings} onChange={setVoiceSettings} sampleText={overlayText} compact />
|
||||
</div>
|
||||
)}
|
||||
</CardContent>
|
||||
|
||||
+18
-101
@@ -25,6 +25,8 @@ import {
|
||||
SelectValue,
|
||||
} from "@/components/ui/select";
|
||||
import { useToast } from "@/hooks/use-toast";
|
||||
import { VoicePicker, type VoiceSettings } from "@/components/reels/voice-picker";
|
||||
import { DEFAULT_TTS_STYLE, DEFAULT_VOICE } from "@shared/voices";
|
||||
import { apiRequest, queryClient, handleUnauthorized, getErrorMessage } from "@/lib/queryClient";
|
||||
import type { SocialPage, Media } from "@shared/schema";
|
||||
import { SiFacebook, SiTiktok } from "react-icons/si";
|
||||
@@ -67,7 +69,8 @@ export default function NewReel() {
|
||||
const [musicVolume, setMusicVolume] = useState([25]);
|
||||
const [ttsEnabled, setTtsEnabled] = useState(true);
|
||||
const [drawText, setDrawText] = useState(true);
|
||||
const [stabilize, setStabilize] = useState(true); // default to true
|
||||
// Désactivée par défaut : double le temps de rendu, utile seulement pour une vidéo tremblée
|
||||
const [stabilize, setStabilize] = useState(false);
|
||||
const [enableEndingEffect, setEnableEndingEffect] = useState(true);
|
||||
|
||||
// TTS Sync state
|
||||
@@ -79,24 +82,13 @@ export default function NewReel() {
|
||||
warnings: string[];
|
||||
} | null>(null);
|
||||
|
||||
// TTS Engine & Voice
|
||||
const [ttsEngine, setTtsEngine] = useState<'edge' | 'gemini'>('edge');
|
||||
const [ttsVoice, setTtsVoice] = useState('fr-FR-VivienneMultilingualNeural');
|
||||
|
||||
// Gemini native TTS voices (Charon = homme, Kore = femme)
|
||||
const geminiVoices = [
|
||||
{ label: 'Charon - Voix Homme', value: 'fr-FR-Standard-B' },
|
||||
{ label: 'Kore - Voix Femme', value: 'fr-FR-Standard-A' },
|
||||
];
|
||||
|
||||
// French Edge TTS voices
|
||||
const edgeVoices = [
|
||||
{ label: 'Vivienne (Femme)', value: 'fr-FR-VivienneMultilingualNeural' },
|
||||
{ label: 'Henri (Homme)', value: 'fr-FR-HenriNeural' },
|
||||
{ label: 'Denise (Femme)', value: 'fr-FR-DeniseNeural' },
|
||||
{ label: 'Rémy (Homme)', value: 'fr-FR-RemyMultilingualNeural' },
|
||||
{ label: 'Jenny (Anglaise, Femme)', value: 'en-US-JennyNeural' },
|
||||
];
|
||||
// Voix : moteur, voix et ton de lecture
|
||||
const [voiceSettings, setVoiceSettings] = useState<VoiceSettings>({
|
||||
engine: 'gemini',
|
||||
voice: DEFAULT_VOICE.gemini,
|
||||
style: DEFAULT_TTS_STYLE,
|
||||
});
|
||||
const { engine: ttsEngine, voice: ttsVoice, style: ttsStyle } = voiceSettings;
|
||||
|
||||
// Enable TTS by default on mobile
|
||||
useEffect(() => {
|
||||
@@ -413,6 +405,7 @@ export default function NewReel() {
|
||||
ttsEnabled,
|
||||
ttsEngine,
|
||||
ttsVoice,
|
||||
ttsStyle,
|
||||
drawText,
|
||||
stabilize: stabilize,
|
||||
enableEndingEffect,
|
||||
@@ -819,88 +812,12 @@ export default function NewReel() {
|
||||
<span>TTS — voix activée</span>
|
||||
</div>
|
||||
|
||||
{/* TTS Engine Selector */}
|
||||
<div className="flex items-center gap-4 mt-3">
|
||||
<Label className="text-sm font-medium">Moteur:</Label>
|
||||
<div className="flex gap-2">
|
||||
<Button
|
||||
size="sm"
|
||||
variant={ttsEngine === 'edge' ? 'default' : 'outline'}
|
||||
onClick={() => {
|
||||
setTtsEngine('edge');
|
||||
setTtsVoice('fr-FR-VivienneMultilingualNeural');
|
||||
}}
|
||||
>
|
||||
Edge TTS
|
||||
</Button>
|
||||
<Button
|
||||
size="sm"
|
||||
variant={ttsEngine === 'gemini' ? 'default' : 'outline'}
|
||||
onClick={() => {
|
||||
setTtsEngine('gemini');
|
||||
setTtsVoice('fr-FR-Standard-B');
|
||||
}}
|
||||
>
|
||||
Gemini TTS
|
||||
</Button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{/* Voice Selector */}
|
||||
<div className="flex items-center gap-3 mt-3">
|
||||
<Label className="text-sm font-medium shrink-0">Voix:</Label>
|
||||
<Select
|
||||
value={ttsVoice}
|
||||
onValueChange={setTtsVoice}
|
||||
>
|
||||
<SelectTrigger className="flex-1">
|
||||
<SelectValue />
|
||||
</SelectTrigger>
|
||||
<SelectContent>
|
||||
{ttsEngine === 'edge' ? (
|
||||
edgeVoices.map(v => (
|
||||
<SelectItem key={v.value} value={v.value}>{v.label}</SelectItem>
|
||||
))
|
||||
) : (
|
||||
geminiVoices.map(v => (
|
||||
<SelectItem key={v.value} value={v.value}>{v.label}</SelectItem>
|
||||
))
|
||||
)}
|
||||
</SelectContent>
|
||||
</Select>
|
||||
</div>
|
||||
|
||||
<div className="mt-3 flex gap-2">
|
||||
<Button
|
||||
size="sm"
|
||||
variant="secondary"
|
||||
className="w-full"
|
||||
onClick={async (e) => {
|
||||
e.stopPropagation();
|
||||
const textToTest = overlayText || "Ceci est un test de voix pour votre vidéo.";
|
||||
try {
|
||||
const response = await apiRequest('POST', '/api/reels/tts-preview', {
|
||||
text: textToTest,
|
||||
ttsEngine,
|
||||
ttsVoice,
|
||||
});
|
||||
const data = await response.json();
|
||||
if (data.success && data.audioBase64) {
|
||||
const audio = new Audio(`data:audio/mp3;base64,${data.audioBase64}`);
|
||||
audio.play();
|
||||
}
|
||||
} catch (err) {
|
||||
toast({
|
||||
title: "Erreur",
|
||||
description: "Impossible de tester la voix",
|
||||
variant: "destructive"
|
||||
});
|
||||
}
|
||||
}}
|
||||
>
|
||||
<Play className="w-3 h-3 mr-2" />
|
||||
Tester la voix
|
||||
</Button>
|
||||
<div className="mt-3">
|
||||
<VoicePicker
|
||||
value={voiceSettings}
|
||||
onChange={setVoiceSettings}
|
||||
sampleText={overlayText}
|
||||
/>
|
||||
</div>
|
||||
|
||||
{syncInfo && (
|
||||
|
||||
@@ -2,6 +2,8 @@ import { useState, useRef } from "react";
|
||||
import Sidebar from "@/components/sidebar";
|
||||
import TopBar from "@/components/topbar";
|
||||
import { useToast } from "@/hooks/use-toast";
|
||||
import { VoicePicker, type VoiceSettings } from "@/components/reels/voice-picker";
|
||||
import { DEFAULT_TTS_STYLE, DEFAULT_VOICE } from "@shared/voices";
|
||||
import { Card, CardContent, CardDescription, CardHeader, CardTitle } from "@/components/ui/card";
|
||||
import { Button } from "@/components/ui/button";
|
||||
import { Textarea } from "@/components/ui/textarea";
|
||||
@@ -34,23 +36,11 @@ export default function RemotionVideoPage() {
|
||||
const [productInfo, setProductInfo] = useState("");
|
||||
const [generatedVariants, setGeneratedVariants] = useState<any[]>([]);
|
||||
const [ttsEnabled, setTtsEnabled] = useState(true);
|
||||
const [ttsEngine, setTtsEngine] = useState<'edge' | 'gemini'>('edge');
|
||||
const [ttsVoice, setTtsVoice] = useState('fr-FR-VivienneMultilingualNeural');
|
||||
|
||||
// Gemini native TTS voices (Charon = homme, Kore = femme)
|
||||
const geminiVoices = [
|
||||
{ label: 'Charon - Voix Homme', value: 'fr-FR-Standard-B' },
|
||||
{ label: 'Kore - Voix Femme', value: 'fr-FR-Standard-A' },
|
||||
];
|
||||
|
||||
// French Edge TTS voices
|
||||
const edgeVoices = [
|
||||
{ label: 'Vivienne (Femme)', value: 'fr-FR-VivienneMultilingualNeural' },
|
||||
{ label: 'Henri (Homme)', value: 'fr-FR-HenriNeural' },
|
||||
{ label: 'Denise (Femme)', value: 'fr-FR-DeniseNeural' },
|
||||
{ label: 'Rémy (Homme)', value: 'fr-FR-RemyMultilingualNeural' },
|
||||
{ label: 'Jenny (Anglaise, Femme)', value: 'en-US-JennyNeural' },
|
||||
];
|
||||
const [voiceSettings, setVoiceSettings] = useState<VoiceSettings>({
|
||||
engine: 'gemini',
|
||||
voice: DEFAULT_VOICE.gemini,
|
||||
style: DEFAULT_TTS_STYLE,
|
||||
});
|
||||
const [musicFile, setMusicFile] = useState<File | null>(null);
|
||||
const [selectedTrack, setSelectedTrack] = useState<AudioTrack | null>(null);
|
||||
const [musicVolume, setMusicVolume] = useState(0.3);
|
||||
@@ -109,16 +99,6 @@ export default function RemotionVideoPage() {
|
||||
});
|
||||
};
|
||||
|
||||
const handleTtsPreview = async () => {
|
||||
const ttsText = overlayText.replace(/#\w+/g, '').replace(/[\uD800-\uDFFF\u2600-\u27BF]/g, '').replace(/\s+/g, ' ').trim();
|
||||
if (!ttsText) return;
|
||||
try {
|
||||
const r = await apiRequest('POST', '/api/reels/tts-preview', { text: ttsText, ttsEngine, ttsVoice });
|
||||
const data = await r.json();
|
||||
if (data.success && data.audioBase64) new window.Audio(`data:audio/mp3;base64,${data.audioBase64}`).play();
|
||||
} catch { toast({ title: "Erreur prévisualisation voix", variant: "destructive" }); }
|
||||
};
|
||||
|
||||
const togglePlayPreview = (track: AudioTrack) => {
|
||||
if (!audioRef.current) return;
|
||||
if (isPlaying === track.id) { audioRef.current.pause(); setIsPlaying(null); }
|
||||
@@ -134,8 +114,10 @@ export default function RemotionVideoPage() {
|
||||
images.forEach(img => formData.append("images", img));
|
||||
selectedLibraryImages.forEach(m => formData.append("existingImageUrls", m.originalUrl));
|
||||
if (overlayText) formData.append("overlayText", overlayText);
|
||||
formData.append("ttsEngine", ttsEngine);
|
||||
formData.append("ttsVoice", ttsVoice);
|
||||
formData.append("ttsEnabled", String(ttsEnabled));
|
||||
formData.append("ttsEngine", voiceSettings.engine);
|
||||
formData.append("ttsVoice", voiceSettings.voice);
|
||||
formData.append("ttsStyle", voiceSettings.style);
|
||||
if (selectedPageIds[0]) formData.append("selectedPageId", selectedPageIds[0]);
|
||||
if (musicFile) { formData.append("music", musicFile); formData.append("musicVolume", String(musicVolume)); }
|
||||
else if (selectedTrack) { formData.append("musicTrackUrl", selectedTrack.url); formData.append("musicVolume", String(musicVolume)); }
|
||||
@@ -294,41 +276,7 @@ export default function RemotionVideoPage() {
|
||||
<div className="space-y-3 p-3 bg-muted/30 rounded-lg border">
|
||||
<p className="text-sm text-muted-foreground">TTS — voix activée</p>
|
||||
|
||||
{/* TTS Engine Selector */}
|
||||
<div className="flex items-center gap-2">
|
||||
<Label className="text-xs font-medium">Moteur:</Label>
|
||||
<div className="flex gap-1">
|
||||
<Button size="sm" variant={ttsEngine === 'edge' ? 'default' : 'outline'} className="h-7 text-xs"
|
||||
onClick={() => { setTtsEngine('edge'); setTtsVoice('fr-FR-VivienneMultilingualNeural'); }}>
|
||||
Edge
|
||||
</Button>
|
||||
<Button size="sm" variant={ttsEngine === 'gemini' ? 'default' : 'outline'} className="h-7 text-xs"
|
||||
onClick={() => { setTtsEngine('gemini'); setTtsVoice('fr-FR-Standard-B'); }}>
|
||||
Gemini
|
||||
</Button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{/* Voice Selector */}
|
||||
<div className="flex items-center gap-2">
|
||||
<Label className="text-xs font-medium shrink-0">Voix:</Label>
|
||||
<Select value={ttsVoice} onValueChange={setTtsVoice}>
|
||||
<SelectTrigger className="h-7 text-xs flex-1">
|
||||
<SelectValue />
|
||||
</SelectTrigger>
|
||||
<SelectContent>
|
||||
{ttsEngine === 'edge' ? (
|
||||
edgeVoices.map(v => <SelectItem key={v.value} value={v.value}>{v.label}</SelectItem>)
|
||||
) : (
|
||||
geminiVoices.map(v => <SelectItem key={v.value} value={v.value}>{v.label}</SelectItem>)
|
||||
)}
|
||||
</SelectContent>
|
||||
</Select>
|
||||
</div>
|
||||
|
||||
<Button size="sm" variant="outline" className="w-full" onClick={handleTtsPreview} disabled={!overlayText}>
|
||||
<Volume2 className="mr-2 w-3 h-3" /> Écouter la voix
|
||||
</Button>
|
||||
<VoicePicker value={voiceSettings} onChange={setVoiceSettings} sampleText={overlayText} />
|
||||
</div>
|
||||
)}
|
||||
</CardContent>
|
||||
|
||||
@@ -71,12 +71,16 @@ services:
|
||||
build:
|
||||
context: ./ffmpeg-service
|
||||
dockerfile: Dockerfile
|
||||
args:
|
||||
# Modèle Whisper (calage des sous-titres) intégré à l'image
|
||||
WHISPER_MODEL: ${WHISPER_MODEL:-base}
|
||||
image: socialflow-ffmpeg-api:latest
|
||||
pull_policy: build
|
||||
container_name: socialflow-ffmpeg
|
||||
restart: unless-stopped
|
||||
environment:
|
||||
API_KEY: ${FFMPEG_API_KEY:-socialflow-secret-ffmpeg-key}
|
||||
GEMINI_TTS_MODEL: ${GEMINI_TTS_MODEL:-gemini-2.5-flash-preview-tts}
|
||||
networks:
|
||||
- internal
|
||||
expose:
|
||||
|
||||
@@ -0,0 +1,4 @@
|
||||
__pycache__
|
||||
.pytest_cache
|
||||
.ruff_cache
|
||||
tests
|
||||
@@ -0,0 +1,3 @@
|
||||
__pycache__/
|
||||
.pytest_cache/
|
||||
.ruff_cache/
|
||||
+23
-22
@@ -1,38 +1,39 @@
|
||||
FROM python:3.11-slim
|
||||
|
||||
# Install system dependencies including FFmpeg with vidstab support and fonts
|
||||
# We need to build/install libvidstab and compile ffmpeg with it, OR use static build
|
||||
RUN apt-get update && apt-get install -y \
|
||||
wget \
|
||||
xz-utils \
|
||||
fonts-dejavu \
|
||||
fontconfig \
|
||||
build-essential \
|
||||
python3-dev \
|
||||
# Polices des sous-titres (Montserrat) et des emojis, installées au build
|
||||
# plutôt que téléchargées à chaque démarrage.
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
wget xz-utils ca-certificates fontconfig \
|
||||
fonts-montserrat fonts-dejavu-core fonts-noto-color-emoji \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# Download static FFmpeg build with all filters including vidstab
|
||||
RUN wget -q https://johnvansickle.com/ffmpeg/releases/ffmpeg-release-amd64-static.tar.xz \
|
||||
&& tar xf ffmpeg-release-amd64-static.tar.xz \
|
||||
&& mv ffmpeg-*-amd64-static/ffmpeg /usr/local/bin/ \
|
||||
&& mv ffmpeg-*-amd64-static/ffprobe /usr/local/bin/ \
|
||||
# FFmpeg statique (vidstab, zimg pour le tone-mapping HDR, libass).
|
||||
# Version épinglée : une mise à jour ne change plus le rendu à notre insu.
|
||||
ARG FFMPEG_VERSION=7.0.2
|
||||
RUN wget -q https://johnvansickle.com/ffmpeg/releases/ffmpeg-${FFMPEG_VERSION}-amd64-static.tar.xz \
|
||||
|| wget -q https://johnvansickle.com/ffmpeg/old-releases/ffmpeg-${FFMPEG_VERSION}-amd64-static.tar.xz \
|
||||
&& tar xf ffmpeg-${FFMPEG_VERSION}-amd64-static.tar.xz \
|
||||
&& mv ffmpeg-*-amd64-static/ffmpeg ffmpeg-*-amd64-static/ffprobe /usr/local/bin/ \
|
||||
&& rm -rf ffmpeg-* \
|
||||
&& fc-cache -fv
|
||||
&& fc-cache -f
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
# Copy requirements first for better caching
|
||||
COPY requirements.txt .
|
||||
RUN pip install --no-cache-dir -r requirements.txt
|
||||
|
||||
# Copy application code
|
||||
# Modèle Whisper (calage des sous-titres) téléchargé au build : aucun accès
|
||||
# réseau nécessaire au premier rendu.
|
||||
ARG WHISPER_MODEL=base
|
||||
ENV WHISPER_MODEL=${WHISPER_MODEL} \
|
||||
HF_HOME=/opt/hf-cache \
|
||||
HOME=/tmp \
|
||||
XDG_CACHE_HOME=/tmp/.cache
|
||||
RUN python -c "from faster_whisper import WhisperModel; WhisperModel('${WHISPER_MODEL}', device='cpu', compute_type='int8')"
|
||||
|
||||
COPY main.py .
|
||||
COPY app ./app
|
||||
|
||||
# Create API Key env var (should be overridden in docker-compose)
|
||||
ENV API_KEY=default-dev-key
|
||||
|
||||
# Expose port
|
||||
EXPOSE 8000
|
||||
|
||||
# Run the application
|
||||
CMD ["uvicorn", "main:app", "--host", "0.0.0.0", "--port", "8000"]
|
||||
Binary file not shown.
@@ -0,0 +1 @@
|
||||
"""Service audio et vidéo des Reels SocialFlow."""
|
||||
@@ -0,0 +1,132 @@
|
||||
"""Calage mot à mot du texte affiché sur la voix.
|
||||
|
||||
Le texte affiché (ponctuation, majuscules) diffère légèrement des mots
|
||||
reconnus ou annoncés par le moteur de voix : on apparie les deux séquences
|
||||
(difflib) et on interpole les mots sans correspondance. Remplace ffsubsync,
|
||||
qui ne faisait qu'un décalage global du texte réparti uniformément.
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import difflib
|
||||
import logging
|
||||
import re
|
||||
import unicodedata
|
||||
from dataclasses import dataclass
|
||||
from functools import lru_cache
|
||||
from pathlib import Path
|
||||
|
||||
from . import config
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@dataclass
|
||||
class Word:
|
||||
text: str
|
||||
start: float
|
||||
end: float
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
return {"text": self.text, "start": round(self.start, 3), "end": round(self.end, 3)}
|
||||
|
||||
|
||||
def normalize(token: str) -> str:
|
||||
"""Forme comparable d'un mot : minuscules, sans accents ni ponctuation."""
|
||||
decomposed = unicodedata.normalize("NFKD", token.lower())
|
||||
stripped = "".join(c for c in decomposed if not unicodedata.combining(c))
|
||||
return re.sub(r"[^a-z0-9]", "", stripped)
|
||||
|
||||
|
||||
def display_tokens(text: str) -> list[str]:
|
||||
"""Mots à afficher ; la ponctuation isolée est rattachée au mot précédent."""
|
||||
tokens: list[str] = []
|
||||
for raw in text.split():
|
||||
if tokens and not normalize(raw):
|
||||
tokens[-1] += raw if raw in ",.!?;:…" else f" {raw}"
|
||||
else:
|
||||
tokens.append(raw)
|
||||
return tokens
|
||||
|
||||
|
||||
def align_words(display_text: str, spoken: list[Word], total_duration: float | None = None) -> list[Word]:
|
||||
"""Attribue à chaque mot affiché un début et une fin tirés des mots prononcés."""
|
||||
tokens = display_tokens(display_text)
|
||||
if not tokens:
|
||||
return []
|
||||
if not spoken:
|
||||
return _spread(tokens, 0.0, total_duration or len(tokens) * 0.4)
|
||||
|
||||
timed: list[Word | None] = [None] * len(tokens)
|
||||
matcher = difflib.SequenceMatcher(
|
||||
a=[normalize(t) for t in tokens], b=[normalize(w.text) for w in spoken], autojunk=False
|
||||
)
|
||||
for block in matcher.get_matching_blocks():
|
||||
for k in range(block.size):
|
||||
src = spoken[block.b + k]
|
||||
timed[block.a + k] = Word(tokens[block.a + k], src.start, src.end)
|
||||
|
||||
# Mots non appariés : répartis dans le trou entre leurs voisins datés
|
||||
end_of_speech = max(spoken[-1].end, total_duration or 0.0)
|
||||
i = 0
|
||||
while i < len(tokens):
|
||||
if timed[i] is not None:
|
||||
i += 1
|
||||
continue
|
||||
j = i
|
||||
while j < len(tokens) and timed[j] is None:
|
||||
j += 1
|
||||
gap_start = timed[i - 1].end if i > 0 else 0.0
|
||||
gap_end = timed[j].start if j < len(tokens) else end_of_speech
|
||||
if gap_end - gap_start < 0.05 * (j - i):
|
||||
gap_end = gap_start + 0.25 * (j - i)
|
||||
timed[i:j] = _spread(tokens[i:j], gap_start, gap_end)
|
||||
i = j
|
||||
|
||||
words = [w for w in timed if w is not None]
|
||||
# Monotonie stricte : un mot ne commence jamais avant la fin du précédent
|
||||
for prev, cur in zip(words, words[1:]):
|
||||
cur.start = max(cur.start, prev.start + 0.01)
|
||||
cur.end = max(cur.end, cur.start + 0.05)
|
||||
return words
|
||||
|
||||
|
||||
def _spread(tokens: list[str], start: float, end: float) -> list[Word]:
|
||||
"""Répartit des mots sur un intervalle au prorata de leur longueur."""
|
||||
weights = [max(1, len(normalize(t))) for t in tokens]
|
||||
total = sum(weights)
|
||||
words, cursor = [], start
|
||||
for token, weight in zip(tokens, weights):
|
||||
duration = (end - start) * weight / total
|
||||
words.append(Word(token, cursor, cursor + duration))
|
||||
cursor += duration
|
||||
return words
|
||||
|
||||
|
||||
@lru_cache(maxsize=1)
|
||||
def _model():
|
||||
from faster_whisper import WhisperModel
|
||||
|
||||
log.info("Chargement du modèle Whisper %s", config.WHISPER_MODEL)
|
||||
return WhisperModel(config.WHISPER_MODEL, device="cpu", compute_type=config.WHISPER_COMPUTE_TYPE)
|
||||
|
||||
|
||||
def _transcribe_sync(audio: Path, hint: str | None) -> list[Word]:
|
||||
segments, _ = _model().transcribe(
|
||||
str(audio),
|
||||
language="fr",
|
||||
word_timestamps=True,
|
||||
initial_prompt=hint or None,
|
||||
vad_filter=False,
|
||||
beam_size=5,
|
||||
)
|
||||
return [
|
||||
Word(w.word.strip(), float(w.start), float(w.end))
|
||||
for segment in segments
|
||||
for w in (segment.words or [])
|
||||
if w.word.strip()
|
||||
]
|
||||
|
||||
|
||||
async def transcribe(audio: Path, hint: str | None = None) -> list[Word]:
|
||||
"""Mots prononcés et leurs instants (Whisper, exécuté hors de la boucle async)."""
|
||||
return await asyncio.to_thread(_transcribe_sync, audio, hint)
|
||||
@@ -0,0 +1,304 @@
|
||||
"""API HTTP du service : voix, rendu des Reels et récupération des fichiers."""
|
||||
|
||||
import asyncio
|
||||
import base64
|
||||
import contextlib
|
||||
import logging
|
||||
import time
|
||||
from pathlib import Path
|
||||
|
||||
import httpx
|
||||
from fastapi import Depends, FastAPI, Header, HTTPException
|
||||
from fastapi.responses import FileResponse
|
||||
from pydantic import BaseModel
|
||||
|
||||
from . import align, config, jobs, proc, render, subtitles, tts, voices
|
||||
from .audio import encode_preview
|
||||
from .text import clean_text
|
||||
|
||||
logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s: %(message)s")
|
||||
log = logging.getLogger("reels")
|
||||
|
||||
if not config.API_KEY:
|
||||
raise RuntimeError("API_KEY doit être défini : le service refuse de démarrer sans clé.")
|
||||
|
||||
# Un seul encodage à la fois : ils saturent le CPU, les suivants attendent
|
||||
render_slot = asyncio.Semaphore(1)
|
||||
|
||||
|
||||
@contextlib.asynccontextmanager
|
||||
async def lifespan(_app: FastAPI):
|
||||
removed = jobs.purge_expired()
|
||||
log.info("Service prêt (%d dossier(s) expiré(s) purgé(s))", removed)
|
||||
|
||||
async def purge_loop():
|
||||
while True:
|
||||
await asyncio.sleep(600)
|
||||
jobs.purge_expired()
|
||||
|
||||
task = asyncio.create_task(purge_loop())
|
||||
yield
|
||||
task.cancel()
|
||||
|
||||
|
||||
app = FastAPI(title="SocialFlow Reels", lifespan=lifespan)
|
||||
|
||||
|
||||
def require_key(x_api_key: str | None = Header(None)) -> None:
|
||||
if x_api_key != config.API_KEY:
|
||||
raise HTTPException(status_code=401, detail="Invalid API Key")
|
||||
|
||||
|
||||
class TtsRequest(BaseModel):
|
||||
text: str
|
||||
tts_voice: str | None = None
|
||||
tts_engine: str | None = "gemini"
|
||||
tts_style: str | None = None
|
||||
gemini_api_key: str | None = None
|
||||
|
||||
|
||||
class ReelRequest(BaseModel):
|
||||
video_base64: str | None = None
|
||||
video_url: str | None = None
|
||||
text: str | None = None
|
||||
music_url: str | None = None
|
||||
watermark_url: str | None = None
|
||||
store_name: str | None = None
|
||||
font_size: int = 64
|
||||
music_volume: float = 0.25
|
||||
tts_enabled: bool = False
|
||||
tts_voice: str | None = None
|
||||
tts_engine: str | None = "gemini"
|
||||
tts_style: str | None = None
|
||||
gemini_api_key: str | None = None
|
||||
draw_text: bool = True
|
||||
stabilize: bool = False
|
||||
enable_ending_effect: bool = True
|
||||
# Champs d'anciennes versions, acceptés et ignorés
|
||||
music_id: str | None = None
|
||||
word_duration: float | None = None
|
||||
|
||||
|
||||
@app.get("/health", dependencies=[Depends(require_key)])
|
||||
async def health():
|
||||
return {"status": "ok", "version": 2}
|
||||
|
||||
|
||||
@app.get("/voices", dependencies=[Depends(require_key)])
|
||||
async def list_voices():
|
||||
return voices.catalog()
|
||||
|
||||
|
||||
@app.post("/preview-tts", dependencies=[Depends(require_key)])
|
||||
async def preview_tts(request: TtsRequest):
|
||||
text = clean_text(request.text)
|
||||
if not text:
|
||||
raise HTTPException(status_code=400, detail="Texte vide après nettoyage (emojis et hashtags retirés)")
|
||||
|
||||
job_id, workdir = jobs.new_job()
|
||||
try:
|
||||
track = await _synthesize(text, request.text, request, workdir)
|
||||
preview = workdir / "preview.mp3"
|
||||
await encode_preview(track.path, preview)
|
||||
return {
|
||||
"success": True,
|
||||
"audio_base64": base64.b64encode(preview.read_bytes()).decode(),
|
||||
"duration": track.duration,
|
||||
"words": [w.to_dict() for w in track.words],
|
||||
"engine": track.engine,
|
||||
"voice": track.voice,
|
||||
"warnings": track.warnings,
|
||||
}
|
||||
finally:
|
||||
jobs.remove_job(job_id)
|
||||
|
||||
|
||||
@app.post("/process-reel", dependencies=[Depends(require_key)])
|
||||
async def process_reel(request: ReelRequest):
|
||||
"""Produit le Reel ; le MP4 se récupère ensuite via GET /files/{job_id}/output.mp4."""
|
||||
job_id, workdir = jobs.new_job()
|
||||
stats: dict[str, float] = {}
|
||||
started = time.monotonic()
|
||||
try:
|
||||
async with render_slot:
|
||||
return await _process(request, job_id, workdir, stats, started)
|
||||
except HTTPException:
|
||||
jobs.remove_job(job_id)
|
||||
raise
|
||||
except proc.CommandError as error:
|
||||
jobs.remove_job(job_id)
|
||||
log.error("Rendu %s en échec : %s", job_id, error)
|
||||
raise HTTPException(status_code=500, detail=str(error)) from error
|
||||
except Exception as error:
|
||||
jobs.remove_job(job_id)
|
||||
log.exception("Rendu %s en échec", job_id)
|
||||
raise HTTPException(status_code=500, detail=f"Erreur de rendu : {error}") from error
|
||||
|
||||
|
||||
@app.get("/files/{job_id}/{name}", dependencies=[Depends(require_key)])
|
||||
async def get_file(job_id: str, name: str):
|
||||
path = jobs.job_file(job_id, name)
|
||||
if not path:
|
||||
raise HTTPException(status_code=404, detail="Fichier introuvable ou expiré")
|
||||
return FileResponse(path, media_type="video/mp4" if name.endswith(".mp4") else None)
|
||||
|
||||
|
||||
@app.delete("/jobs/{job_id}", dependencies=[Depends(require_key)])
|
||||
async def delete_job(job_id: str):
|
||||
jobs.remove_job(job_id)
|
||||
return {"success": True}
|
||||
|
||||
|
||||
# --- Orchestration --------------------------------------------------------
|
||||
|
||||
|
||||
async def _synthesize(text: str, display_source: str | None, request, workdir: Path) -> tts.VoiceTrack:
|
||||
try:
|
||||
return await tts.synthesize(
|
||||
text=text,
|
||||
display_text=clean_text(display_source) or text,
|
||||
engine=request.tts_engine,
|
||||
voice=request.tts_voice,
|
||||
style=request.tts_style,
|
||||
gemini_api_key=request.gemini_api_key,
|
||||
workdir=workdir,
|
||||
)
|
||||
except Exception as error:
|
||||
log.exception("Voix impossible à générer")
|
||||
raise HTTPException(status_code=502, detail=f"La voix n'a pas pu être générée : {error}") from error
|
||||
|
||||
|
||||
async def _download(url: str, target: Path, what: str, required: bool) -> bool:
|
||||
try:
|
||||
async with httpx.AsyncClient(timeout=httpx.Timeout(120, connect=15), follow_redirects=True) as client:
|
||||
async with client.stream("GET", url, headers={"User-Agent": "Mozilla/5.0"}) as response:
|
||||
response.raise_for_status()
|
||||
with target.open("wb") as out:
|
||||
async for chunk in response.aiter_bytes(1 << 20):
|
||||
out.write(chunk)
|
||||
return True
|
||||
except Exception as error:
|
||||
if required:
|
||||
detail = f"Téléchargement de {what} impossible : {error}"
|
||||
raise HTTPException(status_code=400, detail=detail) from error
|
||||
log.warning("Téléchargement de %s impossible, on continue sans : %s", what, error)
|
||||
return False
|
||||
|
||||
|
||||
async def _process(request: ReelRequest, job_id: str, workdir: Path, stats: dict, started: float) -> dict:
|
||||
step = time.monotonic()
|
||||
video = workdir / "input.mp4"
|
||||
if request.video_base64:
|
||||
video.write_bytes(base64.b64decode(request.video_base64))
|
||||
elif request.video_url:
|
||||
await _download(request.video_url, video, "la vidéo", required=True)
|
||||
else:
|
||||
raise HTTPException(status_code=400, detail="Aucune vidéo fournie")
|
||||
|
||||
music = workdir / "music.audio"
|
||||
has_music = bool(request.music_url) and await _download(
|
||||
request.music_url, music, "la musique", required=False
|
||||
)
|
||||
watermark = workdir / "watermark.png"
|
||||
has_watermark = bool(request.watermark_url) and await _download(
|
||||
request.watermark_url, watermark, "le logo", required=False
|
||||
)
|
||||
info = await proc.probe(video)
|
||||
if info.duration <= 0:
|
||||
raise HTTPException(status_code=400, detail="Vidéo illisible (durée nulle)")
|
||||
stats["download"] = time.monotonic() - step
|
||||
|
||||
# --- Voix ---
|
||||
step = time.monotonic()
|
||||
track = None
|
||||
spoken_text = clean_text(request.text)
|
||||
if request.tts_enabled and spoken_text:
|
||||
track = await _synthesize(spoken_text, request.text, request, workdir)
|
||||
stats["tts"] = time.monotonic() - step
|
||||
|
||||
plan = render.RenderPlan(
|
||||
video=video,
|
||||
video_duration=info.duration,
|
||||
output=workdir / "output.mp4",
|
||||
is_hdr=info.is_hdr,
|
||||
music=music if has_music else None,
|
||||
music_volume=request.music_volume,
|
||||
voice=track.path if track else None,
|
||||
voice_duration=track.duration if track else 0.0,
|
||||
watermark=watermark if has_watermark else None,
|
||||
ending_effect=request.enable_ending_effect,
|
||||
keep_original_audio=info.has_audio,
|
||||
)
|
||||
|
||||
# --- Sous-titres ---
|
||||
step = time.monotonic()
|
||||
font_size = max(48, round(request.font_size * 1.4))
|
||||
display = clean_text(request.text)
|
||||
if request.draw_text and display:
|
||||
captions = workdir / "captions.ass"
|
||||
if track:
|
||||
subtitles.write_captions(track.words, captions, offset=plan.voice_delay, font_size=font_size)
|
||||
else:
|
||||
words = await _caption_words_without_voice(display, video, info, plan)
|
||||
subtitles.write_captions(words, captions, offset=0.0, font_size=font_size)
|
||||
plan.captions = captions
|
||||
if request.store_name and request.enable_ending_effect and has_watermark:
|
||||
outro = workdir / "outro.ass"
|
||||
subtitles.write_outro(request.store_name, outro, plan.logo_start, plan.total_duration)
|
||||
plan.outro = outro
|
||||
stats["subtitles"] = time.monotonic() - step
|
||||
|
||||
# --- Stabilisation (1re passe) ---
|
||||
step = time.monotonic()
|
||||
if request.stabilize:
|
||||
transforms = workdir / "transforms.trf"
|
||||
try:
|
||||
await proc.run(render.stabilize_detect_command(video, transforms), timeout=600)
|
||||
plan.stabilize_transforms = transforms
|
||||
except proc.CommandError as error:
|
||||
log.warning("Stabilisation ignorée : %s", error)
|
||||
stats["stabilize"] = time.monotonic() - step
|
||||
|
||||
# --- Encodage ---
|
||||
step = time.monotonic()
|
||||
await proc.run(render.build_command(plan), timeout=1200)
|
||||
stats["encode"] = time.monotonic() - step
|
||||
stats["total"] = time.monotonic() - started
|
||||
|
||||
duration = (await proc.probe(plan.output)).duration
|
||||
log.info(
|
||||
"Rendu %s terminé : %.1f s de vidéo, étapes %s",
|
||||
job_id,
|
||||
duration,
|
||||
{k: round(v, 1) for k, v in stats.items()},
|
||||
)
|
||||
|
||||
# Seul le résultat est conservé jusqu'au téléchargement
|
||||
for entry in workdir.iterdir():
|
||||
if entry != plan.output:
|
||||
entry.unlink(missing_ok=True)
|
||||
|
||||
return {
|
||||
"success": True,
|
||||
"job_id": job_id,
|
||||
"output_path": f"/files/{job_id}/output.mp4",
|
||||
"duration": duration,
|
||||
"tts_engine": track.engine if track else None,
|
||||
"tts_voice": track.voice if track else None,
|
||||
"warnings": track.warnings if track else [],
|
||||
"processing_stats": stats,
|
||||
}
|
||||
|
||||
|
||||
async def _caption_words_without_voice(display: str, video: Path, info, plan: render.RenderPlan):
|
||||
"""Sans voix de synthèse : calage sur la parole de la vidéo si elle en contient,
|
||||
sinon texte réparti sur la durée (hors effet de fin)."""
|
||||
if info.has_audio:
|
||||
try:
|
||||
spoken = await align.transcribe(video, hint=display)
|
||||
if len(spoken) >= max(2, len(display.split()) // 3):
|
||||
return align.align_words(display, spoken, info.duration)
|
||||
except Exception as error: # noqa: BLE001 — on retombe sur la répartition
|
||||
log.warning("Transcription de la vidéo impossible : %s", error)
|
||||
end = plan.logo_start if plan.ending_effect and plan.watermark else plan.total_duration
|
||||
return align.align_words(display, [], max(1.0, end - 0.5))
|
||||
@@ -0,0 +1,60 @@
|
||||
"""Traitement de la voix : filtrage, compression et niveau sonore constant."""
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
from . import proc
|
||||
|
||||
# Coupe les basses inutiles, lisse la dynamique et amène la voix à -16 LUFS :
|
||||
# elle reste intelligible par-dessus la musique sans saturer.
|
||||
VOICE_CHAIN = (
|
||||
"highpass=f=80,"
|
||||
"acompressor=threshold=0.1:ratio=3:attack=5:release=120:makeup=2,"
|
||||
"loudnorm=I=-16:TP=-1.5:LRA=7,"
|
||||
"aresample=48000"
|
||||
)
|
||||
|
||||
|
||||
async def process_voice(source: Path, target: Path) -> None:
|
||||
"""Produit un WAV 48 kHz mono prêt à mixer."""
|
||||
await proc.run(
|
||||
[
|
||||
"ffmpeg",
|
||||
"-y",
|
||||
"-hide_banner",
|
||||
"-loglevel",
|
||||
"error",
|
||||
"-i",
|
||||
str(source),
|
||||
"-af",
|
||||
VOICE_CHAIN,
|
||||
"-ac",
|
||||
"1",
|
||||
"-ar",
|
||||
"48000",
|
||||
"-c:a",
|
||||
"pcm_s16le",
|
||||
str(target),
|
||||
],
|
||||
timeout=120,
|
||||
)
|
||||
|
||||
|
||||
async def encode_preview(source: Path, target: Path) -> None:
|
||||
"""MP3 de bonne qualité pour l'écoute dans le navigateur."""
|
||||
await proc.run(
|
||||
[
|
||||
"ffmpeg",
|
||||
"-y",
|
||||
"-hide_banner",
|
||||
"-loglevel",
|
||||
"error",
|
||||
"-i",
|
||||
str(source),
|
||||
"-c:a",
|
||||
"libmp3lame",
|
||||
"-b:a",
|
||||
"160k",
|
||||
str(target),
|
||||
],
|
||||
timeout=120,
|
||||
)
|
||||
@@ -0,0 +1,32 @@
|
||||
"""Configuration lue dans l'environnement."""
|
||||
|
||||
import os
|
||||
from pathlib import Path
|
||||
|
||||
# Pas de clé par défaut : un service exposé avec une clé connue de tous
|
||||
# n'est pas protégé. Le démarrage échoue si elle manque.
|
||||
API_KEY = os.environ.get("API_KEY", "")
|
||||
|
||||
TEMP_DIR = Path(os.environ.get("TEMP_DIR", "/tmp/ffmpeg_processing"))
|
||||
|
||||
# Durée de conservation des fichiers produits (téléchargés par l'application
|
||||
# puis supprimés ; la purge ne rattrape que les oublis).
|
||||
FILE_TTL_SECONDS = int(os.environ.get("FILE_TTL_SECONDS", "3600"))
|
||||
|
||||
GEMINI_TTS_MODEL = os.environ.get("GEMINI_TTS_MODEL", "gemini-2.5-flash-preview-tts")
|
||||
|
||||
# Modèle Whisper utilisé pour caler les sous-titres sur la voix.
|
||||
WHISPER_MODEL = os.environ.get("WHISPER_MODEL", "base")
|
||||
WHISPER_COMPUTE_TYPE = os.environ.get("WHISPER_COMPUTE_TYPE", "int8")
|
||||
|
||||
# Proxy sortant éventuel pour Edge TTS (aiohttp ne lit pas HTTPS_PROXY seul).
|
||||
OUTBOUND_PROXY = os.environ.get("HTTPS_PROXY") or os.environ.get("https_proxy") or None
|
||||
|
||||
# Rendu
|
||||
WIDTH = 1080
|
||||
HEIGHT = 1920
|
||||
FPS = 30
|
||||
# La voix démarre après ce délai (le temps de capter l'attention).
|
||||
VOICE_DELAY = 2.0
|
||||
|
||||
SUBTITLE_FONT = os.environ.get("SUBTITLE_FONT", "Montserrat")
|
||||
@@ -0,0 +1,44 @@
|
||||
"""Dossiers de travail des rendus et purge des fichiers oubliés."""
|
||||
|
||||
import re
|
||||
import shutil
|
||||
import time
|
||||
import uuid
|
||||
from pathlib import Path
|
||||
|
||||
from . import config
|
||||
|
||||
_SAFE_ID = re.compile(r"^[0-9a-f-]{36}$")
|
||||
_SAFE_NAME = re.compile(r"^[\w.-]+$")
|
||||
|
||||
|
||||
def new_job() -> tuple[str, Path]:
|
||||
job_id = str(uuid.uuid4())
|
||||
workdir = config.TEMP_DIR / job_id
|
||||
workdir.mkdir(parents=True)
|
||||
return job_id, workdir
|
||||
|
||||
|
||||
def job_file(job_id: str, name: str) -> Path | None:
|
||||
"""Fichier d'un job, ou None si l'identifiant ou le nom sont suspects."""
|
||||
if not _SAFE_ID.match(job_id) or not _SAFE_NAME.match(name):
|
||||
return None
|
||||
path = config.TEMP_DIR / job_id / name
|
||||
return path if path.is_file() else None
|
||||
|
||||
|
||||
def remove_job(job_id: str) -> None:
|
||||
if _SAFE_ID.match(job_id):
|
||||
shutil.rmtree(config.TEMP_DIR / job_id, ignore_errors=True)
|
||||
|
||||
|
||||
def purge_expired() -> int:
|
||||
"""Supprime les dossiers plus vieux que FILE_TTL_SECONDS."""
|
||||
config.TEMP_DIR.mkdir(parents=True, exist_ok=True)
|
||||
limit = time.time() - config.FILE_TTL_SECONDS
|
||||
removed = 0
|
||||
for entry in config.TEMP_DIR.iterdir():
|
||||
if entry.is_dir() and entry.stat().st_mtime < limit:
|
||||
shutil.rmtree(entry, ignore_errors=True)
|
||||
removed += 1
|
||||
return removed
|
||||
@@ -0,0 +1,78 @@
|
||||
"""Exécution asynchrone de FFmpeg/ffprobe : le serveur reste disponible
|
||||
pendant un encodage (l'ancien subprocess.run bloquait toute l'API)."""
|
||||
|
||||
import asyncio
|
||||
import json
|
||||
import logging
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class CommandError(RuntimeError):
|
||||
"""Commande externe en échec, avec la fin de sa sortie d'erreur."""
|
||||
|
||||
def __init__(self, cmd: list[str], returncode: int, stderr: str):
|
||||
tail = "\n".join(stderr.strip().splitlines()[-15:])
|
||||
super().__init__(f"{cmd[0]} a échoué (code {returncode}) :\n{tail}")
|
||||
self.returncode = returncode
|
||||
self.stderr = stderr
|
||||
|
||||
|
||||
async def run(cmd: list[str], timeout: float = 900) -> str:
|
||||
"""Lance une commande et renvoie sa sortie standard."""
|
||||
log.debug("exec: %s", " ".join(cmd))
|
||||
process = await asyncio.create_subprocess_exec(
|
||||
*cmd, stdout=asyncio.subprocess.PIPE, stderr=asyncio.subprocess.PIPE
|
||||
)
|
||||
try:
|
||||
stdout, stderr = await asyncio.wait_for(process.communicate(), timeout)
|
||||
except TimeoutError as error:
|
||||
process.kill()
|
||||
await process.wait()
|
||||
raise CommandError(cmd, -1, f"délai de {timeout:.0f} s dépassé") from error
|
||||
if process.returncode != 0:
|
||||
raise CommandError(cmd, process.returncode, stderr.decode(errors="replace"))
|
||||
return stdout.decode(errors="replace")
|
||||
|
||||
|
||||
@dataclass
|
||||
class MediaInfo:
|
||||
duration: float
|
||||
has_audio: bool
|
||||
width: int = 0
|
||||
height: int = 0
|
||||
color_transfer: str = ""
|
||||
|
||||
@property
|
||||
def is_hdr(self) -> bool:
|
||||
# HLG (iPhone) ou PQ (HDR10)
|
||||
return self.color_transfer in ("arib-std-b67", "smpte2084")
|
||||
|
||||
|
||||
async def probe(path: Path) -> MediaInfo:
|
||||
out = await run(
|
||||
[
|
||||
"ffprobe",
|
||||
"-v",
|
||||
"error",
|
||||
"-print_format",
|
||||
"json",
|
||||
"-show_format",
|
||||
"-show_streams",
|
||||
str(path),
|
||||
],
|
||||
timeout=60,
|
||||
)
|
||||
data = json.loads(out)
|
||||
streams = data.get("streams", [])
|
||||
video = next((s for s in streams if s.get("codec_type") == "video"), {})
|
||||
duration = float(data.get("format", {}).get("duration") or video.get("duration") or 0)
|
||||
return MediaInfo(
|
||||
duration=duration,
|
||||
has_audio=any(s.get("codec_type") == "audio" for s in streams),
|
||||
width=int(video.get("width") or 0),
|
||||
height=int(video.get("height") or 0),
|
||||
color_transfer=video.get("color_transfer") or "",
|
||||
)
|
||||
@@ -0,0 +1,217 @@
|
||||
"""Construction de la commande FFmpeg d'un Reel (fonction pure, testable)."""
|
||||
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
|
||||
from . import config
|
||||
from .subtitles import filter_path
|
||||
|
||||
FADE_SECONDS = 2.0
|
||||
LOGO_SECONDS = 5.0
|
||||
# Silence laissé après la dernière phrase avant la fin de la vidéo
|
||||
VOICE_TAIL = 0.8
|
||||
# Durée minimale de l'effet de fin (grand logo + nom du magasin) après la voix
|
||||
OUTRO_MIN = 2.5
|
||||
|
||||
# HDR (HLG/PQ) → SDR BT.709 : sans cela, les vidéos iPhone sortent ternes
|
||||
TONEMAP = (
|
||||
"zscale=t=linear:npl=100,format=gbrpf32le,zscale=p=bt709,"
|
||||
"tonemap=tonemap=hable:desat=0,zscale=t=bt709:m=bt709:r=tv,format=yuv420p"
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
class RenderPlan:
|
||||
video: Path
|
||||
video_duration: float
|
||||
output: Path
|
||||
is_hdr: bool = False
|
||||
stabilize_transforms: Path | None = None
|
||||
music: Path | None = None
|
||||
music_volume: float = 0.25
|
||||
voice: Path | None = None
|
||||
voice_duration: float = 0.0
|
||||
voice_delay: float = config.VOICE_DELAY
|
||||
captions: Path | None = None
|
||||
watermark: Path | None = None
|
||||
outro: Path | None = None # nom du magasin (effet de fin)
|
||||
ending_effect: bool = True
|
||||
keep_original_audio: bool = False
|
||||
|
||||
@property
|
||||
def speech_end(self) -> float:
|
||||
return self.voice_delay + self.voice_duration if self.voice else 0.0
|
||||
|
||||
@property
|
||||
def total_duration(self) -> float:
|
||||
"""La vidéo s'allonge (dernière image figée) si la voix dure plus longtemps."""
|
||||
duration = self.video_duration
|
||||
if self.voice:
|
||||
duration = max(duration, self.speech_end + VOICE_TAIL)
|
||||
if self.ending_effect and self.watermark:
|
||||
duration = max(duration, self.speech_end + OUTRO_MIN)
|
||||
return round(duration, 3)
|
||||
|
||||
@property
|
||||
def logo_start(self) -> float:
|
||||
"""Le grand logo n'arrive qu'une fois la voix terminée."""
|
||||
return max(0.0, self.total_duration - LOGO_SECONDS, self.speech_end)
|
||||
|
||||
@property
|
||||
def freeze_duration(self) -> float:
|
||||
return max(0.0, self.total_duration - self.video_duration)
|
||||
|
||||
|
||||
def build_command(plan: RenderPlan) -> list[str]:
|
||||
total = plan.total_duration
|
||||
logo_start = plan.logo_start
|
||||
fade_start = max(0.0, total - FADE_SECONDS)
|
||||
|
||||
cmd = ["ffmpeg", "-y", "-hide_banner", "-loglevel", "error", "-i", str(plan.video)]
|
||||
index = 1
|
||||
music_idx = voice_idx = wm_idx = None
|
||||
if plan.music:
|
||||
# Musique bouclée : une piste plus courte que la vidéo ne coupe plus le son
|
||||
cmd += ["-stream_loop", "-1", "-i", str(plan.music)]
|
||||
music_idx, index = index, index + 1
|
||||
if plan.voice:
|
||||
cmd += ["-i", str(plan.voice)]
|
||||
voice_idx, index = index, index + 1
|
||||
if plan.watermark:
|
||||
cmd += ["-i", str(plan.watermark)]
|
||||
wm_idx, index = index, index + 1
|
||||
|
||||
graph: list[str] = []
|
||||
|
||||
# --- Vidéo ---
|
||||
chain = []
|
||||
if plan.stabilize_transforms:
|
||||
chain.append(
|
||||
f"vidstabtransform=input={filter_path(plan.stabilize_transforms)}:smoothing=30:relative=1:zoom=5,"
|
||||
"unsharp=5:5:0.6:5:5:0.0"
|
||||
)
|
||||
if plan.is_hdr:
|
||||
chain.append(TONEMAP)
|
||||
chain.append(
|
||||
f"scale={config.WIDTH}:{config.HEIGHT}:force_original_aspect_ratio=increase:flags=lanczos,"
|
||||
f"crop={config.WIDTH}:{config.HEIGHT},setsar=1,fps={config.FPS}"
|
||||
)
|
||||
if plan.freeze_duration > 0:
|
||||
chain.append(f"tpad=stop_mode=clone:stop_duration={plan.freeze_duration:.3f}")
|
||||
if plan.captions:
|
||||
chain.append(f"subtitles='{filter_path(plan.captions)}'")
|
||||
graph.append(f"[0:v]{','.join(chain)}[vbase]")
|
||||
|
||||
current = "vbase"
|
||||
if wm_idx is not None:
|
||||
corner = "W-w-30:H-h-30"
|
||||
if plan.outro and plan.ending_effect:
|
||||
graph.append(f"[{wm_idx}:v]scale=200:-1,split=2[wm_small][wm_big0]")
|
||||
graph.append("[wm_big0]scale=-1:300[wm_big]")
|
||||
graph.append(f"[{current}][wm_small]overlay={corner}:enable='lt(t,{logo_start:.3f})'[vwm1]")
|
||||
graph.append(f"[vwm1][wm_big]overlay=(W-w)/2:(H-h)/2-100:enable='gte(t,{logo_start:.3f})'[vwm2]")
|
||||
current = "vwm2"
|
||||
else:
|
||||
graph.append(f"[{wm_idx}:v]scale=200:-1[wm_small]")
|
||||
graph.append(f"[{current}][wm_small]overlay={corner}[vwm1]")
|
||||
current = "vwm1"
|
||||
if plan.outro and plan.ending_effect:
|
||||
graph.append(f"[{current}]subtitles='{filter_path(plan.outro)}'[vout0]")
|
||||
current = "vout0"
|
||||
|
||||
tail = []
|
||||
if plan.ending_effect:
|
||||
tail.append(f"fade=t=out:st={fade_start:.3f}:d={FADE_SECONDS}")
|
||||
tail.append("format=yuv420p")
|
||||
graph.append(f"[{current}]{','.join(tail)}[vout]")
|
||||
|
||||
# --- Audio ---
|
||||
mix = None
|
||||
if voice_idx is not None:
|
||||
delay_ms = int(plan.voice_delay * 1000)
|
||||
graph.append(f"[{voice_idx}:a]aresample=48000,adelay={delay_ms}:all=1,apad[voice]")
|
||||
if music_idx is not None:
|
||||
graph.append(f"[{music_idx}:a]aresample=48000,volume={plan.music_volume:.3f}[music]")
|
||||
|
||||
if voice_idx is not None and music_idx is not None:
|
||||
# La musique baisse automatiquement quand la voix parle (ducking)
|
||||
graph.append("[voice]asplit=2[vmix][vkey]")
|
||||
graph.append("[music][vkey]sidechaincompress=threshold=0.02:ratio=8:attack=20:release=400[ducked]")
|
||||
graph.append("[ducked][vmix]amix=inputs=2:duration=longest:normalize=0[mix]")
|
||||
mix = "mix"
|
||||
elif voice_idx is not None:
|
||||
mix = "voice"
|
||||
elif music_idx is not None:
|
||||
mix = "music"
|
||||
|
||||
audio_map: list[str] = []
|
||||
if mix:
|
||||
# Niveau final recommandé par les réseaux sociaux (~ -14 LUFS)
|
||||
audio_tail = ["loudnorm=I=-14:TP=-1.5:LRA=11", "aresample=48000"]
|
||||
if plan.ending_effect:
|
||||
audio_tail.append(f"afade=t=out:st={fade_start:.3f}:d={FADE_SECONDS}")
|
||||
graph.append(f"[{mix}]{','.join(audio_tail)}[aout]")
|
||||
audio_map = ["-map", "[aout]"]
|
||||
elif plan.keep_original_audio:
|
||||
audio_map = ["-map", "0:a:0"]
|
||||
|
||||
cmd += ["-filter_complex", ";".join(graph), "-map", "[vout]", *audio_map]
|
||||
cmd += [
|
||||
"-t",
|
||||
f"{total:.3f}",
|
||||
"-c:v",
|
||||
"libx264",
|
||||
"-preset",
|
||||
"medium",
|
||||
"-crf",
|
||||
"19",
|
||||
"-maxrate",
|
||||
"10M",
|
||||
"-bufsize",
|
||||
"20M",
|
||||
"-profile:v",
|
||||
"high",
|
||||
"-level",
|
||||
"4.2",
|
||||
"-g",
|
||||
str(config.FPS * 2),
|
||||
"-keyint_min",
|
||||
str(config.FPS),
|
||||
"-pix_fmt",
|
||||
"yuv420p",
|
||||
"-color_primaries",
|
||||
"bt709",
|
||||
"-color_trc",
|
||||
"bt709",
|
||||
"-colorspace",
|
||||
"bt709",
|
||||
"-c:a",
|
||||
"aac",
|
||||
"-b:a",
|
||||
"192k",
|
||||
"-ar",
|
||||
"48000",
|
||||
"-ac",
|
||||
"2",
|
||||
"-movflags",
|
||||
"+faststart",
|
||||
str(plan.output),
|
||||
]
|
||||
return cmd
|
||||
|
||||
|
||||
def stabilize_detect_command(video: Path, transforms: Path) -> list[str]:
|
||||
return [
|
||||
"ffmpeg",
|
||||
"-y",
|
||||
"-hide_banner",
|
||||
"-loglevel",
|
||||
"error",
|
||||
"-i",
|
||||
str(video),
|
||||
"-vf",
|
||||
f"vidstabdetect=stepsize=32:shakiness=8:accuracy=15:result={filter_path(transforms)}",
|
||||
"-f",
|
||||
"null",
|
||||
"-",
|
||||
]
|
||||
@@ -0,0 +1,128 @@
|
||||
"""Sous-titres ASS : mot à mot, style « Reels » lisible sur tout fond."""
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
from . import config
|
||||
from .align import Word
|
||||
|
||||
# Couleurs ASS au format &HAABBGGRR
|
||||
YELLOW = "&H0000E6FF"
|
||||
WHITE = "&H00FFFFFF"
|
||||
BLACK = "&H00000000"
|
||||
SHADOW = "&H64000000"
|
||||
|
||||
# Au-delà, une ligne déborde ou se lit mal en un coup d'œil
|
||||
MAX_WORDS_PER_LINE = 3
|
||||
MAX_CHARS_PER_LINE = 18
|
||||
# Bas du texte à ~70 % de la hauteur : au-dessus de la légende et des boutons
|
||||
# qu'Instagram et TikTok superposent en bas de l'écran.
|
||||
MARGIN_V = 560
|
||||
POP_MS = 90
|
||||
|
||||
|
||||
def ass_time(seconds: float) -> str:
|
||||
seconds = max(0.0, seconds)
|
||||
centis = int(round(seconds * 100))
|
||||
hours, centis = divmod(centis, 360000)
|
||||
minutes, centis = divmod(centis, 6000)
|
||||
secs, centis = divmod(centis, 100)
|
||||
return f"{hours}:{minutes:02d}:{secs:02d}.{centis:02d}"
|
||||
|
||||
|
||||
def escape(text: str) -> str:
|
||||
return text.replace("\\", "").replace("{", "(").replace("}", ")")
|
||||
|
||||
|
||||
def header(styles: list[str]) -> str:
|
||||
return (
|
||||
"[Script Info]\n"
|
||||
"ScriptType: v4.00+\n"
|
||||
f"PlayResX: {config.WIDTH}\n"
|
||||
f"PlayResY: {config.HEIGHT}\n"
|
||||
"ScaledBorderAndShadow: yes\n"
|
||||
"WrapStyle: 2\n\n"
|
||||
"[V4+ Styles]\n"
|
||||
"Format: Name, Fontname, Fontsize, PrimaryColour, SecondaryColour, OutlineColour, "
|
||||
"BackColour, Bold, Italic, Underline, StrikeOut, ScaleX, ScaleY, Spacing, Angle, "
|
||||
"BorderStyle, Outline, Shadow, Alignment, MarginL, MarginR, MarginV, Encoding\n"
|
||||
+ "".join(f"{s}\n" for s in styles)
|
||||
+ "\n[Events]\n"
|
||||
"Format: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text\n"
|
||||
)
|
||||
|
||||
|
||||
def group_lines(words: list[Word]) -> list[list[Word]]:
|
||||
"""Découpe en lignes courtes, coupées de préférence à la ponctuation."""
|
||||
lines: list[list[Word]] = []
|
||||
current: list[Word] = []
|
||||
for word in words:
|
||||
chars = sum(len(w.text) + 1 for w in current) + len(word.text)
|
||||
if current and (len(current) >= MAX_WORDS_PER_LINE or chars > MAX_CHARS_PER_LINE):
|
||||
lines.append(current)
|
||||
current = []
|
||||
current.append(word)
|
||||
if word.text.endswith((".", "!", "?", ",", ":", ";", "…")):
|
||||
lines.append(current)
|
||||
current = []
|
||||
if current:
|
||||
lines.append(current)
|
||||
return lines
|
||||
|
||||
|
||||
def karaoke_line(line: list[Word], line_end: float) -> str:
|
||||
"""Chaque mot passe du blanc au jaune et « saute » légèrement quand il est dit."""
|
||||
parts = []
|
||||
line_start = line[0].start
|
||||
for index, word in enumerate(line):
|
||||
next_start = line[index + 1].start if index + 1 < len(line) else line_end
|
||||
fill_cs = max(1, int(round((next_start - word.start) * 100)))
|
||||
t0 = int(round((word.start - line_start) * 1000))
|
||||
pop = (
|
||||
f"\\fscx100\\fscy100"
|
||||
f"\\t({t0},{t0 + POP_MS},\\fscx112\\fscy112)"
|
||||
f"\\t({t0 + POP_MS},{t0 + 2 * POP_MS},\\fscx100\\fscy100)"
|
||||
)
|
||||
parts.append(f"{{\\kf{fill_cs}{pop}}}{escape(word.text)}")
|
||||
return " ".join(parts)
|
||||
|
||||
|
||||
def write_captions(
|
||||
words: list[Word], path: Path, *, offset: float, font_size: int, end_time: float | None = None
|
||||
) -> None:
|
||||
"""Écrit le fichier ASS des sous-titres, décalés de `offset` secondes."""
|
||||
style = (
|
||||
f"Style: Caption,{config.SUBTITLE_FONT},{font_size},{YELLOW},{WHITE},{BLACK},{SHADOW},"
|
||||
f"-1,0,0,0,100,100,0,0,1,6,3,2,80,80,{MARGIN_V},1"
|
||||
)
|
||||
lines = group_lines(words)
|
||||
events = []
|
||||
for index, line in enumerate(lines):
|
||||
start = line[0].start
|
||||
natural_end = line[-1].end + 0.25
|
||||
if index + 1 < len(lines):
|
||||
end = min(max(natural_end, line[-1].end), lines[index + 1][0].start)
|
||||
else:
|
||||
end = max(natural_end, end_time - offset if end_time else natural_end)
|
||||
end = max(end, start + 0.3)
|
||||
events.append(
|
||||
f"Dialogue: 0,{ass_time(start + offset)},{ass_time(end + offset)},Caption,,0,0,0,,"
|
||||
f"{karaoke_line(line, end)}"
|
||||
)
|
||||
path.write_text(header([style]) + "\n".join(events) + "\n", encoding="utf-8")
|
||||
|
||||
|
||||
def write_outro(store_name: str, path: Path, start: float, end: float) -> None:
|
||||
"""Nom du magasin sous le logo, en fondu, pendant les dernières secondes."""
|
||||
style = (
|
||||
f"Style: Outro,{config.SUBTITLE_FONT},72,{WHITE},{WHITE},{BLACK},{SHADOW},"
|
||||
"-1,0,0,0,100,100,0,0,1,3,4,2,60,60,700,1"
|
||||
)
|
||||
event = (
|
||||
f"Dialogue: 0,{ass_time(start)},{ass_time(end)},Outro,,0,0,0,,{{\\fad(1200,0)}}{escape(store_name)}"
|
||||
)
|
||||
path.write_text(header([style]) + event + "\n", encoding="utf-8")
|
||||
|
||||
|
||||
def filter_path(path: Path) -> str:
|
||||
"""Chemin utilisable dans un filtre FFmpeg (échappement de : et ')."""
|
||||
return str(path).replace("\\", "/").replace(":", "\\:").replace("'", "\\'")
|
||||
@@ -0,0 +1,20 @@
|
||||
"""Nettoyage du texte avant synthèse et affichage."""
|
||||
|
||||
import re
|
||||
|
||||
import emoji
|
||||
|
||||
_HIDDEN = str.maketrans("", "", "")
|
||||
_HASHTAG = re.compile(r"#[\wÀ-ɏ]+")
|
||||
_URL = re.compile(r"https?://\S+")
|
||||
|
||||
|
||||
def clean_text(text: str | None) -> str:
|
||||
"""Retire emojis, hashtags, liens et caractères invisibles."""
|
||||
if not text:
|
||||
return ""
|
||||
text = text.translate(_HIDDEN)
|
||||
text = emoji.replace_emoji(text, replace="")
|
||||
text = _URL.sub("", text)
|
||||
text = _HASHTAG.sub("", text)
|
||||
return " ".join(text.split())
|
||||
@@ -0,0 +1,63 @@
|
||||
"""Synthèse vocale : moteur au choix, repli sur Edge, voix traitée et mots minutés."""
|
||||
|
||||
import logging
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
|
||||
from .. import align, audio, proc
|
||||
from ..align import Word
|
||||
from . import edge, gemini
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@dataclass
|
||||
class VoiceTrack:
|
||||
path: Path # WAV 48 kHz traité
|
||||
duration: float
|
||||
words: list[Word] # mots affichés, minutés depuis le début de la voix
|
||||
engine: str
|
||||
voice: str
|
||||
warnings: list[str] = field(default_factory=list)
|
||||
|
||||
|
||||
async def synthesize(
|
||||
*,
|
||||
text: str,
|
||||
display_text: str,
|
||||
engine: str | None,
|
||||
voice: str | None,
|
||||
style: str | None,
|
||||
gemini_api_key: str | None,
|
||||
workdir: Path,
|
||||
) -> VoiceTrack:
|
||||
warnings: list[str] = []
|
||||
raw: Path | None = None
|
||||
spoken: list[Word] = []
|
||||
used_engine, used_voice = "edge", ""
|
||||
|
||||
if (engine or "gemini") == "gemini":
|
||||
if not gemini_api_key:
|
||||
warnings.append("Clé Gemini absente : voix Edge utilisée à la place.")
|
||||
else:
|
||||
try:
|
||||
raw, used_voice = await gemini.synthesize(text, voice, style, gemini_api_key, workdir)
|
||||
used_engine = "gemini"
|
||||
except Exception as error: # noqa: BLE001 — repli sur Edge
|
||||
log.warning("Gemini TTS en échec, repli sur Edge : %s", error)
|
||||
warnings.append(f"Gemini indisponible ({error}) : voix Edge utilisée à la place.")
|
||||
|
||||
if raw is None:
|
||||
raw, spoken, used_voice = await edge.synthesize(text, voice, style, workdir)
|
||||
|
||||
processed = workdir / "voice.wav"
|
||||
await audio.process_voice(raw, processed)
|
||||
duration = (await proc.probe(processed)).duration
|
||||
|
||||
if not spoken:
|
||||
# Gemini ne donne pas le minutage : Whisper le retrouve dans l'audio
|
||||
spoken = await align.transcribe(processed, hint=text)
|
||||
|
||||
words = align.align_words(display_text, spoken, duration)
|
||||
log.info("Voix prête : %s/%s, %.1f s, %d mots", used_engine, used_voice, duration, len(words))
|
||||
return VoiceTrack(processed, duration, words, used_engine, used_voice, warnings)
|
||||
@@ -0,0 +1,58 @@
|
||||
"""Synthèse vocale Edge (gratuite, fournit le minutage des mots)."""
|
||||
|
||||
import logging
|
||||
from pathlib import Path
|
||||
|
||||
import edge_tts
|
||||
|
||||
from .. import config
|
||||
from ..align import Word
|
||||
from ..voices import edge_fallbacks, resolve_edge_voice
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
# Réglages de lecture par style (Edge n'interprète pas de consigne en texte)
|
||||
STYLE_PROSODY = {
|
||||
"dynamic": ("+8%", "+2Hz"),
|
||||
"promo": ("+10%", "+3Hz"),
|
||||
"calm": ("-8%", "-2Hz"),
|
||||
"warm": ("-2%", "+0Hz"),
|
||||
}
|
||||
|
||||
|
||||
async def synthesize(
|
||||
text: str, voice: str | None, style: str | None, workdir: Path
|
||||
) -> tuple[Path, list[Word], str]:
|
||||
"""Génère la voix ; renvoie le MP3, les mots minutés et la voix utilisée."""
|
||||
rate, pitch = STYLE_PROSODY.get(style or "", ("+0%", "+0Hz"))
|
||||
last_error: Exception | None = None
|
||||
|
||||
for candidate in edge_fallbacks(resolve_edge_voice(voice)):
|
||||
target = workdir / "edge.mp3"
|
||||
try:
|
||||
communicate = edge_tts.Communicate(
|
||||
text,
|
||||
candidate.id,
|
||||
rate=rate,
|
||||
pitch=pitch,
|
||||
# edge-tts 7 ne renvoie plus que les phrases par défaut
|
||||
boundary="WordBoundary",
|
||||
proxy=config.OUTBOUND_PROXY,
|
||||
)
|
||||
words: list[Word] = []
|
||||
with target.open("wb") as out:
|
||||
async for chunk in communicate.stream():
|
||||
if chunk["type"] == "audio":
|
||||
out.write(chunk["data"])
|
||||
elif chunk["type"] == "WordBoundary":
|
||||
start = chunk["offset"] / 10_000_000
|
||||
words.append(Word(chunk["text"], start, start + chunk["duration"] / 10_000_000))
|
||||
if target.stat().st_size == 0:
|
||||
raise RuntimeError("audio vide")
|
||||
log.info("Edge TTS : voix=%s, %d mots minutés", candidate.id, len(words))
|
||||
return target, words, candidate.id
|
||||
except Exception as error: # noqa: BLE001 — on essaie la voix suivante
|
||||
log.warning("Edge TTS %s en échec : %s", candidate.id, error)
|
||||
last_error = error
|
||||
|
||||
raise RuntimeError(f"Aucune voix Edge disponible : {last_error}")
|
||||
@@ -0,0 +1,102 @@
|
||||
"""Synthèse vocale Gemini (API generateContent, modalité audio)."""
|
||||
|
||||
import base64
|
||||
import logging
|
||||
import re
|
||||
from pathlib import Path
|
||||
|
||||
import httpx
|
||||
|
||||
from .. import config, proc
|
||||
from ..voices import resolve_gemini_voice, style_instruction
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
API_ROOT = "https://generativelanguage.googleapis.com/v1beta/models"
|
||||
FALLBACK_MODEL = "gemini-2.5-flash-preview-tts"
|
||||
|
||||
|
||||
class GeminiError(RuntimeError):
|
||||
pass
|
||||
|
||||
|
||||
def build_prompt(text: str, style: str | None) -> str:
|
||||
"""Texte envoyé au modèle : la consigne de style précède le texte à lire."""
|
||||
instruction = style_instruction(style)
|
||||
return f"{instruction} :\n{text}" if instruction else text
|
||||
|
||||
|
||||
async def _request(model: str, prompt: str, voice: str, api_key: str) -> dict:
|
||||
payload = {
|
||||
"contents": [{"parts": [{"text": prompt}]}],
|
||||
"generationConfig": {
|
||||
"responseModalities": ["AUDIO"],
|
||||
"speechConfig": {"voiceConfig": {"prebuiltVoiceConfig": {"voiceName": voice}}},
|
||||
},
|
||||
}
|
||||
async with httpx.AsyncClient(timeout=120) as client:
|
||||
# Clé en en-tête : dans l'URL, elle finissait dans les journaux
|
||||
response = await client.post(
|
||||
f"{API_ROOT}/{model}:generateContent",
|
||||
json=payload,
|
||||
headers={"x-goog-api-key": api_key},
|
||||
)
|
||||
if response.status_code != 200:
|
||||
raise GeminiError(f"Gemini {model} : HTTP {response.status_code} {response.text[:300]}")
|
||||
return response.json()
|
||||
|
||||
|
||||
async def synthesize(
|
||||
text: str, voice: str | None, style: str | None, api_key: str, workdir: Path
|
||||
) -> tuple[Path, str]:
|
||||
"""Génère la voix ; renvoie un WAV brut et le nom de la voix utilisée."""
|
||||
gemini_voice = resolve_gemini_voice(voice).id
|
||||
prompt = build_prompt(text, style)
|
||||
model = config.GEMINI_TTS_MODEL
|
||||
log.info("Gemini TTS : modèle=%s voix=%s style=%s (%d car.)", model, gemini_voice, style, len(text))
|
||||
|
||||
try:
|
||||
data = await _request(model, prompt, gemini_voice, api_key)
|
||||
except GeminiError as error:
|
||||
if model == FALLBACK_MODEL or "HTTP 404" not in str(error):
|
||||
raise
|
||||
log.warning("Modèle %s indisponible, repli sur %s", model, FALLBACK_MODEL)
|
||||
data = await _request(FALLBACK_MODEL, prompt, gemini_voice, api_key)
|
||||
|
||||
try:
|
||||
part = next(p for p in data["candidates"][0]["content"]["parts"] if "inlineData" in p)["inlineData"]
|
||||
except (KeyError, IndexError, StopIteration) as error:
|
||||
raise GeminiError(f"Réponse Gemini sans audio : {str(data)[:300]}") from error
|
||||
|
||||
audio = base64.b64decode(part["data"])
|
||||
mime = part.get("mimeType", "")
|
||||
raw = workdir / "gemini_raw.wav"
|
||||
|
||||
if "wav" in mime:
|
||||
raw.write_bytes(audio)
|
||||
else:
|
||||
# PCM brut 16 bits (« audio/L16;codec=pcm;rate=24000 »)
|
||||
rate_match = re.search(r"rate=(\d+)", mime)
|
||||
pcm = workdir / "gemini.pcm"
|
||||
pcm.write_bytes(audio)
|
||||
await proc.run(
|
||||
[
|
||||
"ffmpeg",
|
||||
"-y",
|
||||
"-hide_banner",
|
||||
"-loglevel",
|
||||
"error",
|
||||
"-f",
|
||||
"s16le",
|
||||
"-ar",
|
||||
rate_match.group(1) if rate_match else "24000",
|
||||
"-ac",
|
||||
"1",
|
||||
"-i",
|
||||
str(pcm),
|
||||
str(raw),
|
||||
],
|
||||
timeout=60,
|
||||
)
|
||||
pcm.unlink(missing_ok=True)
|
||||
return raw, gemini_voice
|
||||
@@ -0,0 +1,131 @@
|
||||
"""Catalogue des voix et des styles de lecture."""
|
||||
|
||||
from dataclasses import dataclass
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Voice:
|
||||
id: str
|
||||
label: str
|
||||
gender: str # "female" | "male"
|
||||
|
||||
|
||||
# Voix natives de Gemini TTS (toutes multilingues, françaises comprises).
|
||||
GEMINI_VOICES: tuple[Voice, ...] = (
|
||||
Voice("Kore", "Kore — ferme", "female"),
|
||||
Voice("Aoede", "Aoede — légère", "female"),
|
||||
Voice("Leda", "Leda — jeune", "female"),
|
||||
Voice("Zephyr", "Zephyr — lumineuse", "female"),
|
||||
Voice("Callirrhoe", "Callirrhoe — décontractée", "female"),
|
||||
Voice("Autonoe", "Autonoe — lumineuse", "female"),
|
||||
Voice("Despina", "Despina — douce", "female"),
|
||||
Voice("Erinome", "Erinome — claire", "female"),
|
||||
Voice("Laomedeia", "Laomedeia — enjouée", "female"),
|
||||
Voice("Achernar", "Achernar — tendre", "female"),
|
||||
Voice("Gacrux", "Gacrux — mûre", "female"),
|
||||
Voice("Pulcherrima", "Pulcherrima — affirmée", "female"),
|
||||
Voice("Vindemiatrix", "Vindemiatrix — délicate", "female"),
|
||||
Voice("Sulafat", "Sulafat — chaleureuse", "female"),
|
||||
Voice("Charon", "Charon — informative", "male"),
|
||||
Voice("Puck", "Puck — enjouée", "male"),
|
||||
Voice("Fenrir", "Fenrir — enthousiaste", "male"),
|
||||
Voice("Orus", "Orus — ferme", "male"),
|
||||
Voice("Enceladus", "Enceladus — soufflée", "male"),
|
||||
Voice("Iapetus", "Iapetus — claire", "male"),
|
||||
Voice("Umbriel", "Umbriel — décontractée", "male"),
|
||||
Voice("Algieba", "Algieba — veloutée", "male"),
|
||||
Voice("Algenib", "Algenib — rocailleuse", "male"),
|
||||
Voice("Rasalgethi", "Rasalgethi — informative", "male"),
|
||||
Voice("Alnilam", "Alnilam — ferme", "male"),
|
||||
Voice("Schedar", "Schedar — posée", "male"),
|
||||
Voice("Achird", "Achird — amicale", "male"),
|
||||
Voice("Zubenelgenubi", "Zubenelgenubi — naturelle", "male"),
|
||||
Voice("Sadachbia", "Sadachbia — vive", "male"),
|
||||
Voice("Sadaltager", "Sadaltager — experte", "male"),
|
||||
)
|
||||
|
||||
EDGE_VOICES: tuple[Voice, ...] = (
|
||||
Voice("fr-FR-VivienneMultilingualNeural", "Vivienne", "female"),
|
||||
Voice("fr-FR-DeniseNeural", "Denise", "female"),
|
||||
Voice("fr-FR-EloiseNeural", "Eloise", "female"),
|
||||
Voice("fr-FR-RemyMultilingualNeural", "Rémy", "male"),
|
||||
Voice("fr-FR-HenriNeural", "Henri", "male"),
|
||||
)
|
||||
|
||||
# Consignes de lecture ajoutées au texte envoyé à Gemini.
|
||||
STYLES: dict[str, tuple[str, str]] = {
|
||||
"neutral": ("Neutre", ""),
|
||||
"dynamic": (
|
||||
"Dynamique",
|
||||
"Lis ce texte en français sur un ton dynamique et enthousiaste, avec un rythme "
|
||||
"entraînant, comme une vidéo courte sur les réseaux sociaux",
|
||||
),
|
||||
"warm": (
|
||||
"Chaleureux",
|
||||
"Lis ce texte en français sur un ton chaleureux, souriant et proche, comme si tu conseillais un ami",
|
||||
),
|
||||
"calm": (
|
||||
"Calme",
|
||||
"Lis ce texte en français sur un ton calme, posé et rassurant, sans te presser",
|
||||
),
|
||||
"promo": (
|
||||
"Promo",
|
||||
"Lis ce texte en français comme une annonce promotionnelle énergique, en "
|
||||
"insistant sur les offres et les prix",
|
||||
),
|
||||
}
|
||||
|
||||
_GEMINI_BY_ID = {v.id.lower(): v for v in GEMINI_VOICES}
|
||||
_EDGE_BY_ID = {v.id: v for v in EDGE_VOICES}
|
||||
|
||||
|
||||
def resolve_gemini_voice(voice: str | None) -> Voice:
|
||||
"""Voix Gemini à utiliser.
|
||||
|
||||
Accepte aussi les anciens identifiants « fr-FR-Standard-X » encore stockés
|
||||
dans des jobs (B/D = voix masculine).
|
||||
"""
|
||||
if voice and voice.lower() in _GEMINI_BY_ID:
|
||||
return _GEMINI_BY_ID[voice.lower()]
|
||||
if voice and voice.endswith(("-B", "-D")) or voice == "male":
|
||||
return _GEMINI_BY_ID["charon"]
|
||||
return _GEMINI_BY_ID["kore"]
|
||||
|
||||
|
||||
def voice_gender(voice: str | None) -> str:
|
||||
"""Genre d'une voix, quel que soit le moteur d'origine."""
|
||||
if not voice:
|
||||
return "female"
|
||||
if voice.lower() in _GEMINI_BY_ID:
|
||||
return _GEMINI_BY_ID[voice.lower()].gender
|
||||
if voice in _EDGE_BY_ID:
|
||||
return _EDGE_BY_ID[voice].gender
|
||||
if voice == "male" or voice.endswith(("-B", "-D")):
|
||||
return "male"
|
||||
return "male" if any(name in voice for name in ("Remy", "Henri", "Paul")) else "female"
|
||||
|
||||
|
||||
def resolve_edge_voice(voice: str | None) -> Voice:
|
||||
"""Voix Edge à utiliser ; une voix Gemini est remplacée par une voix Edge du même genre."""
|
||||
if voice in _EDGE_BY_ID:
|
||||
return _EDGE_BY_ID[voice]
|
||||
gender = voice_gender(voice)
|
||||
return next(v for v in EDGE_VOICES if v.gender == gender)
|
||||
|
||||
|
||||
def edge_fallbacks(primary: Voice) -> list[Voice]:
|
||||
"""Voix françaises de secours du même genre (jamais d'anglais)."""
|
||||
same = [v for v in EDGE_VOICES if v.gender == primary.gender and v != primary]
|
||||
return [primary, *same]
|
||||
|
||||
|
||||
def style_instruction(style: str | None) -> str:
|
||||
return STYLES.get(style or "neutral", STYLES["neutral"])[1]
|
||||
|
||||
|
||||
def catalog() -> dict:
|
||||
return {
|
||||
"gemini": [v.__dict__ for v in GEMINI_VOICES],
|
||||
"edge": [v.__dict__ for v in EDGE_VOICES],
|
||||
"styles": [{"id": k, "label": v[0]} for k, v in STYLES.items()],
|
||||
}
|
||||
+5
-1385
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,3 @@
|
||||
-r requirements.txt
|
||||
pytest==8.3.5
|
||||
ruff==0.11.8
|
||||
@@ -1,9 +1,7 @@
|
||||
fastapi==0.109.0
|
||||
uvicorn==0.27.0
|
||||
python-multipart==0.0.6
|
||||
requests==2.31.0
|
||||
pydantic==2.6.0
|
||||
edge-tts==6.1.12
|
||||
emoji
|
||||
ffsubsync==0.4.26
|
||||
httpx>=0.25.0
|
||||
fastapi==0.115.12
|
||||
uvicorn[standard]==0.34.2
|
||||
pydantic==2.11.4
|
||||
httpx==0.28.1
|
||||
edge-tts==7.2.8
|
||||
emoji==2.14.1
|
||||
faster-whisper==1.2.1
|
||||
@@ -0,0 +1,9 @@
|
||||
line-length = 110
|
||||
target-version = "py311"
|
||||
|
||||
[lint]
|
||||
select = ["E", "F", "W", "I", "B", "UP"]
|
||||
ignore = [
|
||||
"B008", # Depends() dans les signatures FastAPI
|
||||
"B905", # zip() sur des séquences dont les longueurs sont liées par construction
|
||||
]
|
||||
@@ -1,14 +0,0 @@
|
||||
[Script Info]
|
||||
ScriptType: v4.00+
|
||||
PlayResX: 1080
|
||||
PlayResY: 1920
|
||||
ScaledBorderAndShadow: yes
|
||||
|
||||
[V4+ Styles]
|
||||
Format: Name, Fontname, Fontsize, PrimaryColour, SecondaryColour, OutlineColour, BackColour, Bold, Italic, Underline, StrikeOut, ScaleX, ScaleY, Spacing, Angle, BorderStyle, Outline, Shadow, Alignment, MarginL, MarginR, MarginV, Encoding
|
||||
Style: Default,Sans,65,&H0000FFFF,&H00FFFFFF,&H00000000,&H80000000,-1,0,0,0,100,100,0,0,1,4,2,5,50,50,0,1
|
||||
|
||||
[Events]
|
||||
Format: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text
|
||||
Dialogue: 0,0:00:02.00,0:00:02.71,Default,,0,0,0,,{\kf31}Test {\kf31}test {\kf10}1
|
||||
Dialogue: 0,0:00:02.71,0:00:02.87,Default,,0,0,0,,{\kf10}2 {\kf10}3
|
||||
@@ -1,13 +0,0 @@
|
||||
[Script Info]
|
||||
ScriptType: v4.00+
|
||||
PlayResX: 1080
|
||||
PlayResY: 1920
|
||||
ScaledBorderAndShadow: yes
|
||||
|
||||
[V4+ Styles]
|
||||
Format: Name, Fontname, Fontsize, PrimaryColour, SecondaryColour, OutlineColour, BackColour, Bold, Italic, Underline, StrikeOut, ScaleX, ScaleY, Spacing, Angle, BorderStyle, Outline, Shadow, Alignment, MarginL, MarginR, MarginV, Encoding
|
||||
Style: Default,Sans,65,&H00FFFFFF,&H000000FF,&H00000000,&H80000000,-1,0,0,0,100,100,0,0,1,4,2,5,50,50,0,1
|
||||
|
||||
[Events]
|
||||
Format: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text
|
||||
Dialogue: 0,0:00:02.00,0:00:04.00,Default,,0,0,0,,Hello World
|
||||
Whitespace-only changes.
@@ -0,0 +1,3 @@
|
||||
import os
|
||||
|
||||
os.environ.setdefault("API_KEY", "test-key")
|
||||
@@ -0,0 +1,31 @@
|
||||
from app.align import Word, align_words, display_tokens, normalize
|
||||
|
||||
|
||||
def test_normalize_strips_accents_and_punctuation():
|
||||
assert normalize("Été,") == "ete"
|
||||
assert normalize("«") == ""
|
||||
|
||||
|
||||
def test_display_tokens_attach_isolated_punctuation():
|
||||
assert display_tokens("Bonjour ! Venez vite .") == ["Bonjour!", "Venez", "vite."]
|
||||
|
||||
|
||||
def test_align_uses_engine_timings_and_keeps_display_text():
|
||||
spoken = [Word("Bonjour", 0.05, 0.6), Word("découvrez", 0.99, 1.35), Word("nos", 1.35, 1.44)]
|
||||
words = align_words("Bonjour, découvrez nos", spoken)
|
||||
assert [w.text for w in words] == ["Bonjour,", "découvrez", "nos"]
|
||||
assert words[1].start == 0.99
|
||||
|
||||
|
||||
def test_unmatched_words_are_interpolated_between_neighbours():
|
||||
spoken = [Word("les", 0.0, 0.2), Word("prix", 1.0, 1.3)]
|
||||
words = align_words("les super promos prix", spoken)
|
||||
assert [w.text for w in words] == ["les", "super", "promos", "prix"]
|
||||
assert 0.2 <= words[1].start < words[2].start < 1.0
|
||||
assert all(b.start > a.start for a, b in zip(words, words[1:]))
|
||||
|
||||
|
||||
def test_without_spoken_words_text_is_spread_over_duration():
|
||||
words = align_words("un deux trois", [], total_duration=3.0)
|
||||
assert words[0].start == 0.0
|
||||
assert abs(words[-1].end - 3.0) < 1e-6
|
||||
@@ -0,0 +1,77 @@
|
||||
from pathlib import Path
|
||||
|
||||
from app.render import RenderPlan, build_command
|
||||
|
||||
|
||||
def _graph(cmd):
|
||||
return cmd[cmd.index("-filter_complex") + 1]
|
||||
|
||||
|
||||
def test_voice_longer_than_video_extends_with_frozen_frame():
|
||||
plan = RenderPlan(
|
||||
video=Path("in.mp4"),
|
||||
video_duration=5.0,
|
||||
output=Path("out.mp4"),
|
||||
voice=Path("v.wav"),
|
||||
voice_duration=10.0,
|
||||
)
|
||||
assert plan.total_duration == 12.8
|
||||
cmd = build_command(plan)
|
||||
assert "tpad=stop_mode=clone:stop_duration=7.800" in _graph(cmd)
|
||||
assert cmd[cmd.index("-t") + 1] == "12.800"
|
||||
|
||||
|
||||
def test_music_is_looped_and_ducked_under_voice():
|
||||
plan = RenderPlan(
|
||||
video=Path("in.mp4"),
|
||||
video_duration=20.0,
|
||||
output=Path("out.mp4"),
|
||||
music=Path("m.mp3"),
|
||||
voice=Path("v.wav"),
|
||||
voice_duration=5.0,
|
||||
)
|
||||
cmd = build_command(plan)
|
||||
assert cmd[cmd.index("m.mp3") - 3 : cmd.index("m.mp3")] == ["-stream_loop", "-1", "-i"]
|
||||
graph = _graph(cmd)
|
||||
assert "sidechaincompress" in graph
|
||||
assert "amix=inputs=2:duration=longest" in graph
|
||||
assert "loudnorm=I=-14" in graph
|
||||
|
||||
|
||||
def test_no_brightness_hack_and_hdr_tonemapped():
|
||||
plan = RenderPlan(video=Path("in.mp4"), video_duration=5.0, output=Path("out.mp4"), is_hdr=True)
|
||||
graph = _graph(build_command(plan))
|
||||
assert "eq=" not in graph
|
||||
assert "tonemap=tonemap=hable" in graph
|
||||
assert "flags=lanczos" in graph
|
||||
|
||||
|
||||
def test_original_audio_kept_when_nothing_added():
|
||||
plan = RenderPlan(
|
||||
video=Path("in.mp4"), video_duration=5.0, output=Path("out.mp4"), keep_original_audio=True
|
||||
)
|
||||
cmd = build_command(plan)
|
||||
assert cmd[cmd.index("-map", cmd.index("[vout]")) + 1] == "0:a:0"
|
||||
|
||||
|
||||
def test_encoding_targets_social_networks():
|
||||
cmd = build_command(RenderPlan(video=Path("in.mp4"), video_duration=5.0, output=Path("out.mp4")))
|
||||
joined = " ".join(cmd)
|
||||
assert "-crf 19" in joined and "-b:v" not in joined
|
||||
assert "-ar 48000" in joined and "-b:a 192k" in joined
|
||||
assert "-colorspace bt709" in joined
|
||||
|
||||
|
||||
def test_big_logo_waits_for_the_end_of_the_voice():
|
||||
plan = RenderPlan(
|
||||
video=Path("in.mp4"),
|
||||
video_duration=6.0,
|
||||
output=Path("out.mp4"),
|
||||
voice=Path("v.wav"),
|
||||
voice_duration=4.8,
|
||||
watermark=Path("logo.png"),
|
||||
outro=Path("o.ass"),
|
||||
)
|
||||
assert plan.logo_start == 6.8
|
||||
assert plan.total_duration == 9.3
|
||||
assert "gte(t,6.800)" in _graph(build_command(plan))
|
||||
@@ -0,0 +1,23 @@
|
||||
from app.align import Word
|
||||
from app.subtitles import ass_time, group_lines, write_captions
|
||||
|
||||
|
||||
def test_ass_time_format():
|
||||
assert ass_time(0) == "0:00:00.00"
|
||||
assert ass_time(61.234) == "0:01:01.23"
|
||||
|
||||
|
||||
def test_lines_break_on_punctuation_and_length():
|
||||
words = [Word(t, i, i + 0.5) for i, t in enumerate("Bonjour, venez découvrir nos nouveautés".split())]
|
||||
lines = group_lines(words)
|
||||
assert [w.text for w in lines[0]] == ["Bonjour,"]
|
||||
assert all(len(line) <= 3 for line in lines)
|
||||
|
||||
|
||||
def test_captions_are_offset_and_escape_braces(tmp_path):
|
||||
path = tmp_path / "c.ass"
|
||||
write_captions([Word("{prix}", 0.0, 0.5), Word("fous", 0.5, 1.0)], path, offset=2.0, font_size=80)
|
||||
content = path.read_text()
|
||||
assert "Dialogue: 0,0:00:02.00," in content
|
||||
assert "{prix}" not in content.split("[Events]")[1].replace("{\\", "")
|
||||
assert "(prix)" in content
|
||||
@@ -0,0 +1,23 @@
|
||||
from app.tts.gemini import build_prompt
|
||||
from app.voices import GEMINI_VOICES, edge_fallbacks, resolve_edge_voice, resolve_gemini_voice
|
||||
|
||||
|
||||
def test_thirty_gemini_voices():
|
||||
assert len(GEMINI_VOICES) == 30
|
||||
|
||||
|
||||
def test_legacy_gemini_identifiers_still_resolve():
|
||||
assert resolve_gemini_voice("fr-FR-Standard-B").id == "Charon"
|
||||
assert resolve_gemini_voice("fr-FR-Standard-A").id == "Kore"
|
||||
assert resolve_gemini_voice("puck").id == "Puck"
|
||||
|
||||
|
||||
def test_edge_fallbacks_are_french_and_same_gender():
|
||||
voice = resolve_edge_voice("Fenrir") # voix Gemini masculine
|
||||
assert voice.gender == "male"
|
||||
assert all(v.id.startswith("fr-FR") and v.gender == "male" for v in edge_fallbacks(voice))
|
||||
|
||||
|
||||
def test_style_prompt_precedes_text():
|
||||
assert build_prompt("Salut", "neutral") == "Salut"
|
||||
assert build_prompt("Salut", "dynamic").endswith(":\nSalut")
|
||||
+32
-102
@@ -5,14 +5,13 @@
|
||||
import { Router, Request, Response } from 'express';
|
||||
import type { User } from '@shared/schema';
|
||||
import { storage } from '../storage';
|
||||
import { ffmpegService } from '../services/ffmpeg';
|
||||
import { resolveInternalUrl } from '../services/minio';
|
||||
import { videoReelParamsSchema, type VideoReelParams } from '@shared/reel';
|
||||
import { ffmpegService, FFmpegServiceError } from '../services/ffmpeg';
|
||||
import { ttsPreviewSchema, videoReelParamsSchema, type VideoReelParams } from '@shared/reel';
|
||||
import { enqueueReelJob, countActiveReelJobs } from '../services/reels/queue';
|
||||
import { resolveGeminiApiKey, resolveLogoPath, resolveMusicUrl, resolveStoreName } from '../services/reels/assets';
|
||||
import { resolveGeminiApiKey, resolveStoreName } from '../services/reels/assets';
|
||||
import { openRouterService, describeGenerationError } from '../services/openrouter';
|
||||
|
||||
import { ttsSyncService } from '../services/ttsSync';
|
||||
import { estimateVoiceTiming } from '../services/ttsSync';
|
||||
/** Piste musicale telle qu'attendue par le client. */
|
||||
interface MusicTrack {
|
||||
id: string;
|
||||
@@ -270,105 +269,37 @@ reelsRouter.post('/reels/generate-text', async (req: Request, res: Response) =>
|
||||
}
|
||||
});
|
||||
|
||||
/**
|
||||
* Prévisualiser un Reel (traitement sans publication)
|
||||
* POST /api/reels/preview
|
||||
*/
|
||||
reelsRouter.post('/reels/preview', async (req: Request, res: Response) => {
|
||||
try {
|
||||
const user = req.user as User;
|
||||
const {
|
||||
videoMediaId,
|
||||
musicTrackId,
|
||||
musicUrl,
|
||||
overlayText,
|
||||
ttsEnabled,
|
||||
ttsVoice,
|
||||
ttsEngine,
|
||||
wordDuration = 0.6,
|
||||
fontSize = 64,
|
||||
musicVolume = 0.25,
|
||||
drawText = true,
|
||||
stabilize = false,
|
||||
enableEndingEffect = true,
|
||||
} = req.body;
|
||||
|
||||
// Récupérer le média vidéo
|
||||
const media = await storage.getMediaById(videoMediaId);
|
||||
if (!media) {
|
||||
return res.status(404).json({ error: 'Vidéo non trouvée' });
|
||||
}
|
||||
|
||||
if (media.type !== 'video') {
|
||||
return res.status(400).json({ error: 'Le média doit être une vidéo' });
|
||||
}
|
||||
|
||||
const [finalMusicUrl, logoPath, geminiApiKey] = await Promise.all([
|
||||
resolveMusicUrl(musicTrackId, musicUrl),
|
||||
resolveLogoPath(),
|
||||
resolveGeminiApiKey(ttsEngine),
|
||||
]);
|
||||
const watermarkUrl = logoPath ? resolveInternalUrl(logoPath) : undefined;
|
||||
|
||||
const finalWordDuration = wordDuration;
|
||||
|
||||
// Traiter la vidéo via FFmpeg
|
||||
const result = await ffmpegService.processReelFromUrl(resolveInternalUrl(media.originalUrl), {
|
||||
text: overlayText,
|
||||
musicUrl: finalMusicUrl,
|
||||
ttsEnabled,
|
||||
ttsVoice,
|
||||
ttsEngine,
|
||||
geminiApiKey,
|
||||
wordDuration: finalWordDuration,
|
||||
fontSize,
|
||||
musicVolume,
|
||||
drawText,
|
||||
stabilize,
|
||||
watermarkUrl,
|
||||
enableEndingEffect,
|
||||
});
|
||||
|
||||
if (!result.success) {
|
||||
return res.status(500).json({ error: result.error || 'Erreur de traitement vidéo' });
|
||||
}
|
||||
|
||||
// Retourner la vidéo en base64 pour prévisualisation
|
||||
res.json({
|
||||
success: true,
|
||||
videoBase64: result.videoBase64,
|
||||
duration: result.duration,
|
||||
});
|
||||
} catch (error) {
|
||||
console.error('❌ Error previewing Reel:', error);
|
||||
res.status(500).json({ error: 'Erreur lors de la prévisualisation du Reel' });
|
||||
}
|
||||
});
|
||||
|
||||
/**
|
||||
* Prévisualiser la voix TTS
|
||||
* POST /api/reels/tts-preview
|
||||
*/
|
||||
reelsRouter.post('/reels/tts-preview', async (req: Request, res: Response) => {
|
||||
const parsed = ttsPreviewSchema.safeParse(req.body);
|
||||
if (!parsed.success) {
|
||||
return res.status(400).json({ error: parsed.error.issues[0]?.message ?? 'Paramètres invalides' });
|
||||
}
|
||||
const { text, ttsVoice, ttsEngine, ttsStyle } = parsed.data;
|
||||
|
||||
try {
|
||||
const user = req.user as User;
|
||||
const { text, ttsVoice, ttsEngine } = req.body;
|
||||
|
||||
if (!text) {
|
||||
return res.status(400).json({ error: 'Texte requis' });
|
||||
}
|
||||
|
||||
const geminiApiKey = await resolveGeminiApiKey(ttsEngine);
|
||||
const result = await ffmpegService.previewTTS(text, ttsVoice, ttsEngine, geminiApiKey);
|
||||
|
||||
if (!result.success) {
|
||||
return res.status(500).json({ error: result.error || 'Erreur de génération TTS' });
|
||||
}
|
||||
|
||||
res.json({ success: true, audioBase64: result.audioBase64 });
|
||||
const preview = await ffmpegService.previewVoice(text, {
|
||||
voice: ttsVoice,
|
||||
engine: ttsEngine,
|
||||
style: ttsStyle,
|
||||
geminiApiKey: await resolveGeminiApiKey(ttsEngine),
|
||||
});
|
||||
res.json({
|
||||
success: true,
|
||||
audioBase64: preview.audio.toString('base64'),
|
||||
duration: preview.duration,
|
||||
words: preview.words,
|
||||
engine: preview.engine,
|
||||
voice: preview.voice,
|
||||
warnings: preview.warnings,
|
||||
});
|
||||
} catch (error) {
|
||||
console.error('❌ Error generating TTS preview:', error);
|
||||
res.status(500).json({ error: 'Erreur lors de la génération de la voix' });
|
||||
const message = error instanceof Error ? error.message : 'Erreur lors de la génération de la voix';
|
||||
res.status(error instanceof FFmpegServiceError && error.status === 400 ? 400 : 502).json({ error: message });
|
||||
}
|
||||
});
|
||||
|
||||
@@ -378,13 +309,12 @@ reelsRouter.post('/reels/tts-preview', async (req: Request, res: Response) => {
|
||||
*/
|
||||
reelsRouter.post('/reels/sync-info', async (req: Request, res: Response) => {
|
||||
try {
|
||||
const { text, ttsVoice, ttsEngine } = req.body;
|
||||
if (!text || !ttsVoice) {
|
||||
return res.status(400).json({ error: 'Texte et voix requis' });
|
||||
const { text } = req.body;
|
||||
if (!text) {
|
||||
return res.status(400).json({ error: 'Texte requis' });
|
||||
}
|
||||
const geminiApiKey = await resolveGeminiApiKey(ttsEngine);
|
||||
const sync = await ttsSyncService.calculateSyncTiming(text, ttsVoice, ttsEngine, geminiApiKey);
|
||||
res.json(sync);
|
||||
// Estimation locale : ne déclenche aucune synthèse (payante avec Gemini)
|
||||
res.json(estimateVoiceTiming(text));
|
||||
} catch (error) {
|
||||
console.error('❌ Error calculating sync:', error);
|
||||
res.status(500).json({ error: 'Erreur de calcul de synchronisation' });
|
||||
|
||||
@@ -70,8 +70,10 @@ remotionRouter.post("/render", upload.fields([{ name: "images", maxCount: 4 }, {
|
||||
overlayText: req.body.overlayText,
|
||||
musicUrl: musicFile ? `/uploads/temp/${path.basename(musicFile.path)}` : req.body.musicTrackUrl,
|
||||
musicVolume: Number.isFinite(musicVolume) ? musicVolume : undefined,
|
||||
ttsEnabled: req.body.ttsEnabled !== "false",
|
||||
ttsEngine: req.body.ttsEngine || undefined,
|
||||
ttsVoice: req.body.ttsVoice,
|
||||
ttsStyle: req.body.ttsStyle || undefined,
|
||||
storeName: await resolveStoreName(user.id, req.body.selectedPageId),
|
||||
tempFiles: uploaded.map((f) => f.path),
|
||||
});
|
||||
|
||||
+144
-289
@@ -1,214 +1,120 @@
|
||||
/**
|
||||
* FFmpeg Docker API Service
|
||||
*
|
||||
* Intégration avec l'API FFmpeg Docker locale pour le traitement vidéo des Reels.
|
||||
* L'API attend une vidéo en base64 et retourne la vidéo traitée en base64.
|
||||
* Client du service FFmpeg (conteneur Python `ffmpeg-service`).
|
||||
*
|
||||
* Le service rend le Reel puis expose le MP4 en téléchargement
|
||||
* (GET /files/{job}/output.mp4) : la vidéo ne transite plus en base64 dans du
|
||||
* JSON, qui gonflait sa taille d'un tiers et la gardait entière en mémoire.
|
||||
*/
|
||||
|
||||
interface FFmpegReelRequest {
|
||||
video_base64?: string; // Vidéo source en base64
|
||||
video_url?: string; // OU URL de la vidéo source
|
||||
text?: string; // Texte overlay style TikTok
|
||||
music_id?: string; // ID de la musique (catalogue FFmpeg)
|
||||
music_url?: string; // OU URL directe de la musique
|
||||
tts_enabled?: boolean; // Activation du TTS
|
||||
tts_voice?: string; // Voix TTS (ex: fr-FR-VivienneNeural)
|
||||
tts_engine?: string; // Moteur TTS: "edge" ou "gemini"
|
||||
gemini_api_key?: string; // Clé API Google Gemini pour TTS
|
||||
word_duration?: number; // Durée par mot (default: 0.6s)
|
||||
font_size?: number; // Taille police (default: 24)
|
||||
music_volume?: number; // Volume musique (default: 0.25)
|
||||
draw_text?: boolean; // Dessiner le texte sur la vidéo (default: true)
|
||||
stabilize?: boolean; // Stabilisation vidéo via vidstab (default: false)
|
||||
watermark_url?: string; // URL du logo
|
||||
store_name?: string; // Nom du magasin pour l'outro
|
||||
enable_ending_effect?: boolean; // Activer l'effet de fin (logo+fondu)
|
||||
}
|
||||
|
||||
interface FFmpegReelResponse {
|
||||
success: boolean;
|
||||
output_base64?: string;
|
||||
duration?: number;
|
||||
detail?: string;
|
||||
tts_error?: string;
|
||||
}
|
||||
import type { TtsEngine, TtsStyle } from '@shared/voices';
|
||||
|
||||
/** Un rendu long (stabilisation + encodage) peut dépasser plusieurs minutes. */
|
||||
const PROCESS_TIMEOUT_MS = 15 * 60_000;
|
||||
const TTS_TIMEOUT_MS = 2 * 60_000;
|
||||
const PROCESS_TIMEOUT_MS = 20 * 60_000;
|
||||
const DOWNLOAD_TIMEOUT_MS = 5 * 60_000;
|
||||
const TTS_TIMEOUT_MS = 3 * 60_000;
|
||||
const HEALTH_TIMEOUT_MS = 5_000;
|
||||
|
||||
export interface ReelRenderOptions {
|
||||
text?: string;
|
||||
musicUrl?: string;
|
||||
ttsEnabled?: boolean;
|
||||
ttsVoice?: string;
|
||||
ttsEngine?: TtsEngine;
|
||||
ttsStyle?: TtsStyle;
|
||||
geminiApiKey?: string;
|
||||
fontSize?: number;
|
||||
musicVolume?: number;
|
||||
drawText?: boolean;
|
||||
stabilize?: boolean;
|
||||
watermarkUrl?: string;
|
||||
storeName?: string;
|
||||
enableEndingEffect?: boolean;
|
||||
}
|
||||
|
||||
export interface ReelRenderResult {
|
||||
video: Buffer;
|
||||
duration: number;
|
||||
ttsEngine?: string | null;
|
||||
ttsVoice?: string | null;
|
||||
warnings: string[];
|
||||
}
|
||||
|
||||
export interface TimedWord {
|
||||
text: string;
|
||||
start: number;
|
||||
end: number;
|
||||
}
|
||||
|
||||
export interface VoicePreview {
|
||||
audio: Buffer;
|
||||
duration: number;
|
||||
words: TimedWord[];
|
||||
engine: string;
|
||||
voice: string;
|
||||
warnings: string[];
|
||||
}
|
||||
|
||||
interface FFmpegConfig {
|
||||
apiUrl: string;
|
||||
apiKey: string;
|
||||
}
|
||||
|
||||
/** Erreur renvoyée par le service, avec son message lisible. */
|
||||
export class FFmpegServiceError extends Error {
|
||||
constructor(message: string, readonly status?: number) {
|
||||
super(message);
|
||||
this.name = 'FFmpegServiceError';
|
||||
}
|
||||
}
|
||||
|
||||
export class FFmpegService {
|
||||
private config: FFmpegConfig | null = null;
|
||||
|
||||
/**
|
||||
* Configure le service avec l'URL et la clé API
|
||||
*/
|
||||
configure(apiUrl: string, apiKey: string): void {
|
||||
this.config = { apiUrl, apiKey };
|
||||
console.log('🎬 FFmpeg Service configured:', apiUrl);
|
||||
this.config = { apiUrl: apiUrl.replace(/\/$/, ''), apiKey };
|
||||
console.log('🎬 FFmpeg Service configured:', this.config.apiUrl);
|
||||
}
|
||||
|
||||
/**
|
||||
* Vérifie que le service est configuré
|
||||
*/
|
||||
private ensureConfigured(): FFmpegConfig {
|
||||
if (!this.config) {
|
||||
throw new Error('FFmpeg Service not configured. Call configure() first.');
|
||||
throw new FFmpegServiceError("Le service FFmpeg n'est pas configuré (FFMPEG_API_URL / FFMPEG_API_KEY).");
|
||||
}
|
||||
return this.config;
|
||||
}
|
||||
|
||||
/**
|
||||
* Traite une vidéo pour créer un Reel avec musique et texte overlay
|
||||
*
|
||||
* @param videoBase64 - Vidéo source encodée en base64
|
||||
* @param options - Options de traitement (texte, musique, etc.)
|
||||
* @returns Vidéo traitée en base64
|
||||
*/
|
||||
async processReelVideo(
|
||||
videoBase64: string,
|
||||
options: {
|
||||
text?: string;
|
||||
musicId?: string;
|
||||
musicUrl?: string;
|
||||
ttsEnabled?: boolean;
|
||||
ttsVoice?: string;
|
||||
ttsEngine?: string;
|
||||
geminiApiKey?: string;
|
||||
wordDuration?: number;
|
||||
fontSize?: number;
|
||||
musicVolume?: number;
|
||||
drawText?: boolean;
|
||||
stabilize?: boolean;
|
||||
watermarkUrl?: string;
|
||||
storeName?: string;
|
||||
enableEndingEffect?: boolean;
|
||||
} = {}
|
||||
): Promise<{ success: boolean; videoBase64?: string; duration?: number; error?: string }> {
|
||||
private async call(path: string, init: RequestInit & { timeoutMs: number }): Promise<Response> {
|
||||
const config = this.ensureConfigured();
|
||||
|
||||
const requestBody: FFmpegReelRequest = {
|
||||
video_base64: videoBase64,
|
||||
text: options.text,
|
||||
music_id: options.musicId,
|
||||
music_url: options.musicUrl,
|
||||
tts_enabled: options.ttsEnabled,
|
||||
tts_voice: options.ttsVoice,
|
||||
tts_engine: options.ttsEngine,
|
||||
gemini_api_key: options.geminiApiKey,
|
||||
word_duration: options.wordDuration ?? 0.6,
|
||||
font_size: options.fontSize ?? 64,
|
||||
music_volume: options.musicVolume ?? 0.25,
|
||||
draw_text: options.drawText ?? true,
|
||||
stabilize: options.stabilize ?? false,
|
||||
watermark_url: options.watermarkUrl,
|
||||
store_name: options.storeName,
|
||||
enable_ending_effect: options.enableEndingEffect ?? true,
|
||||
};
|
||||
|
||||
// Remove undefined values
|
||||
Object.keys(requestBody).forEach(key => {
|
||||
if (requestBody[key as keyof FFmpegReelRequest] === undefined) {
|
||||
delete requestBody[key as keyof FFmpegReelRequest];
|
||||
}
|
||||
const { timeoutMs, headers, ...rest } = init;
|
||||
const response = await fetch(`${config.apiUrl}${path}`, {
|
||||
...rest,
|
||||
headers: { 'X-API-Key': config.apiKey, ...headers },
|
||||
signal: AbortSignal.timeout(timeoutMs),
|
||||
});
|
||||
|
||||
console.log('🎬 Processing Reel video:', {
|
||||
hasVideo: !!videoBase64,
|
||||
hasText: !!options.text,
|
||||
hasMusicId: !!options.musicId,
|
||||
hasMusicUrl: !!options.musicUrl,
|
||||
hasTTS: options.ttsEnabled,
|
||||
drawText: options.drawText,
|
||||
});
|
||||
|
||||
try {
|
||||
const response = await fetch(`${config.apiUrl}/process-reel`, {
|
||||
method: 'POST',
|
||||
headers: {
|
||||
'Content-Type': 'application/json',
|
||||
'X-API-Key': config.apiKey,
|
||||
},
|
||||
body: JSON.stringify(requestBody),
|
||||
signal: AbortSignal.timeout(PROCESS_TIMEOUT_MS),
|
||||
});
|
||||
|
||||
if (!response.ok) {
|
||||
const errorText = await response.text();
|
||||
console.error('❌ FFmpeg API error:', response.status, errorText);
|
||||
return {
|
||||
success: false,
|
||||
error: `FFmpeg API error: ${response.status} - ${errorText}`,
|
||||
};
|
||||
}
|
||||
|
||||
const data = await response.json() as FFmpegReelResponse;
|
||||
|
||||
if (!data.success) {
|
||||
console.error('❌ FFmpeg processing failed:', data.detail);
|
||||
return {
|
||||
success: false,
|
||||
error: data.detail || 'Unknown FFmpeg processing error',
|
||||
};
|
||||
}
|
||||
|
||||
console.log('✅ Reel video processed successfully, duration:', data.duration);
|
||||
return {
|
||||
success: true,
|
||||
videoBase64: data.output_base64,
|
||||
duration: data.duration,
|
||||
};
|
||||
|
||||
} catch (error) {
|
||||
console.error('❌ FFmpeg Service error:', error);
|
||||
return {
|
||||
success: false,
|
||||
error: error instanceof Error ? error.message : 'Unknown error',
|
||||
};
|
||||
if (!response.ok) {
|
||||
const body = await response.text();
|
||||
let detail = body;
|
||||
try {
|
||||
detail = JSON.parse(body).detail ?? body;
|
||||
} catch { /* corps non JSON */ }
|
||||
throw new FFmpegServiceError(String(detail).slice(0, 2000), response.status);
|
||||
}
|
||||
return response;
|
||||
}
|
||||
|
||||
/**
|
||||
* Traite une vidéo depuis une URL (télécharge, traite, retourne base64)
|
||||
* Rend un Reel à partir de l'URL d'une vidéo et renvoie le MP4 produit.
|
||||
* Lève FFmpegServiceError en cas d'échec (voix comprise).
|
||||
*/
|
||||
async processReelFromUrl(
|
||||
videoUrl: string,
|
||||
options: {
|
||||
text?: string;
|
||||
musicId?: string;
|
||||
musicUrl?: string;
|
||||
ttsEnabled?: boolean;
|
||||
ttsVoice?: string;
|
||||
ttsEngine?: string;
|
||||
geminiApiKey?: string;
|
||||
wordDuration?: number;
|
||||
fontSize?: number;
|
||||
musicVolume?: number;
|
||||
drawText?: boolean;
|
||||
stabilize?: boolean;
|
||||
watermarkUrl?: string;
|
||||
storeName?: string;
|
||||
enableEndingEffect?: boolean;
|
||||
} = {}
|
||||
): Promise<{ success: boolean; videoBase64?: string; duration?: number; error?: string; ttsError?: string }> {
|
||||
const config = this.ensureConfigured();
|
||||
|
||||
const requestBody: FFmpegReelRequest = {
|
||||
async renderReel(videoUrl: string, options: ReelRenderOptions = {}): Promise<ReelRenderResult> {
|
||||
const body = {
|
||||
video_url: videoUrl,
|
||||
text: options.text,
|
||||
music_id: options.musicId,
|
||||
music_url: options.musicUrl,
|
||||
tts_enabled: options.ttsEnabled,
|
||||
tts_enabled: options.ttsEnabled ?? false,
|
||||
tts_voice: options.ttsVoice,
|
||||
tts_engine: options.ttsEngine,
|
||||
tts_style: options.ttsStyle,
|
||||
gemini_api_key: options.geminiApiKey,
|
||||
word_duration: options.wordDuration ?? 0.6,
|
||||
font_size: options.fontSize ?? 64,
|
||||
music_volume: options.musicVolume ?? 0.25,
|
||||
draw_text: options.drawText ?? true,
|
||||
@@ -218,139 +124,88 @@ export class FFmpegService {
|
||||
enable_ending_effect: options.enableEndingEffect ?? true,
|
||||
};
|
||||
|
||||
// Remove undefined values
|
||||
Object.keys(requestBody).forEach(key => {
|
||||
if (requestBody[key as keyof FFmpegReelRequest] === undefined) {
|
||||
delete requestBody[key as keyof FFmpegReelRequest];
|
||||
}
|
||||
});
|
||||
|
||||
console.log('🎬 Processing Reel from URL:', {
|
||||
console.log('🎬 Rendu du Reel :', {
|
||||
videoUrl,
|
||||
hasText: !!options.text,
|
||||
textLength: options.text?.length || 0,
|
||||
hasMusicId: !!options.musicId,
|
||||
hasMusicUrl: !!options.musicUrl,
|
||||
ttsEnabled: options.ttsEnabled,
|
||||
drawText: options.drawText,
|
||||
textLength: options.text?.length ?? 0,
|
||||
music: !!options.musicUrl,
|
||||
tts: options.ttsEnabled ? `${options.ttsEngine}/${options.ttsVoice}/${options.ttsStyle ?? 'neutral'}` : false,
|
||||
});
|
||||
|
||||
const debugBody = { ...requestBody };
|
||||
console.log('📤 Sending to FFmpeg API:', JSON.stringify({ ...debugBody, text: debugBody.text ? `[${debugBody.text.length} chars]` : undefined }));
|
||||
const response = await this.call('/process-reel', {
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify(body),
|
||||
timeoutMs: PROCESS_TIMEOUT_MS,
|
||||
});
|
||||
const data = await response.json() as {
|
||||
job_id: string;
|
||||
output_path: string;
|
||||
duration: number;
|
||||
tts_engine?: string | null;
|
||||
tts_voice?: string | null;
|
||||
warnings?: string[];
|
||||
};
|
||||
|
||||
try {
|
||||
const response = await fetch(`${config.apiUrl}/process-reel`, {
|
||||
method: 'POST',
|
||||
headers: {
|
||||
'Content-Type': 'application/json',
|
||||
'X-API-Key': config.apiKey,
|
||||
},
|
||||
body: JSON.stringify(requestBody),
|
||||
signal: AbortSignal.timeout(PROCESS_TIMEOUT_MS),
|
||||
});
|
||||
|
||||
if (!response.ok) {
|
||||
const errorText = await response.text();
|
||||
console.error('❌ FFmpeg API error:', response.status, errorText);
|
||||
return {
|
||||
success: false,
|
||||
error: `FFmpeg API error: ${response.status} - ${errorText}`,
|
||||
};
|
||||
}
|
||||
|
||||
const data = await response.json() as FFmpegReelResponse;
|
||||
|
||||
if (!data.success) {
|
||||
console.error('❌ FFmpeg processing failed:', data.detail);
|
||||
return {
|
||||
success: false,
|
||||
error: data.detail || 'Unknown FFmpeg processing error',
|
||||
};
|
||||
}
|
||||
|
||||
if (data.tts_error) {
|
||||
console.error('❌ TTS failed in Python service:', data.tts_error);
|
||||
}
|
||||
console.log('✅ Reel video processed successfully from URL');
|
||||
const file = await this.call(data.output_path, { method: 'GET', timeoutMs: DOWNLOAD_TIMEOUT_MS });
|
||||
const video = Buffer.from(await file.arrayBuffer());
|
||||
for (const warning of data.warnings ?? []) console.warn(`⚠️ [FFmpeg] ${warning}`);
|
||||
return {
|
||||
success: true,
|
||||
videoBase64: data.output_base64,
|
||||
video,
|
||||
duration: data.duration,
|
||||
ttsError: data.tts_error,
|
||||
};
|
||||
|
||||
} catch (error) {
|
||||
console.error('❌ FFmpeg Service error:', error);
|
||||
return {
|
||||
success: false,
|
||||
error: error instanceof Error ? error.message : 'Unknown error',
|
||||
ttsEngine: data.tts_engine,
|
||||
ttsVoice: data.tts_voice,
|
||||
warnings: data.warnings ?? [],
|
||||
};
|
||||
} finally {
|
||||
// Le fichier n'est plus utile au service une fois récupéré
|
||||
this.call(`/jobs/${data.job_id}`, { method: 'DELETE', timeoutMs: HEALTH_TIMEOUT_MS })
|
||||
.catch(() => { /* purgé plus tard par le service */ });
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Vérifie la santé de l'API FFmpeg
|
||||
*/
|
||||
async healthCheck(): Promise<boolean> {
|
||||
try {
|
||||
const config = this.ensureConfigured();
|
||||
const response = await fetch(`${config.apiUrl}/health`, {
|
||||
method: 'GET',
|
||||
headers: {
|
||||
'X-API-Key': config.apiKey,
|
||||
},
|
||||
signal: AbortSignal.timeout(HEALTH_TIMEOUT_MS),
|
||||
});
|
||||
return response.ok;
|
||||
await this.call('/health', { method: 'GET', timeoutMs: HEALTH_TIMEOUT_MS });
|
||||
return true;
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
async previewTTS(
|
||||
|
||||
/** Génère la voix seule (aperçu), avec le minutage de chaque mot. */
|
||||
async previewVoice(
|
||||
text: string,
|
||||
ttsVoice?: string,
|
||||
ttsEngine?: string,
|
||||
geminiApiKey?: string
|
||||
): Promise<{ success: boolean; audioBase64?: string; error?: string }> {
|
||||
const config = this.ensureConfigured();
|
||||
|
||||
try {
|
||||
const response = await fetch(`${config.apiUrl}/preview-tts`, {
|
||||
method: 'POST',
|
||||
headers: {
|
||||
'Content-Type': 'application/json',
|
||||
'X-API-Key': config.apiKey,
|
||||
},
|
||||
body: JSON.stringify({
|
||||
text,
|
||||
tts_enabled: true,
|
||||
tts_voice: ttsVoice,
|
||||
tts_engine: ttsEngine,
|
||||
gemini_api_key: geminiApiKey,
|
||||
}),
|
||||
signal: AbortSignal.timeout(TTS_TIMEOUT_MS),
|
||||
});
|
||||
|
||||
if (!response.ok) {
|
||||
const errorText = await response.text();
|
||||
return { success: false, error: `FFmpeg API error: ${response.status} - ${errorText}` };
|
||||
}
|
||||
|
||||
const data = await response.json();
|
||||
|
||||
if (!data.success) {
|
||||
return { success: false, error: data.detail };
|
||||
}
|
||||
|
||||
return { success: true, audioBase64: data.audio_base64 };
|
||||
|
||||
} catch (error) {
|
||||
console.error('❌ TTS Preview error:', error);
|
||||
return {
|
||||
success: false,
|
||||
error: error instanceof Error ? error.message : 'Unknown error',
|
||||
};
|
||||
}
|
||||
options: { voice?: string; engine?: TtsEngine; style?: TtsStyle; geminiApiKey?: string } = {},
|
||||
): Promise<VoicePreview> {
|
||||
const response = await this.call('/preview-tts', {
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify({
|
||||
text,
|
||||
tts_voice: options.voice,
|
||||
tts_engine: options.engine,
|
||||
tts_style: options.style,
|
||||
gemini_api_key: options.geminiApiKey,
|
||||
}),
|
||||
timeoutMs: TTS_TIMEOUT_MS,
|
||||
});
|
||||
const data = await response.json() as {
|
||||
audio_base64: string;
|
||||
duration: number;
|
||||
words: TimedWord[];
|
||||
engine: string;
|
||||
voice: string;
|
||||
warnings?: string[];
|
||||
};
|
||||
return {
|
||||
audio: Buffer.from(data.audio_base64, 'base64'),
|
||||
duration: data.duration,
|
||||
words: data.words ?? [],
|
||||
engine: data.engine,
|
||||
voice: data.voice,
|
||||
warnings: data.warnings ?? [],
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -6,11 +6,10 @@
|
||||
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import * as musicMetadata from "music-metadata";
|
||||
import { bundle } from "@remotion/bundler";
|
||||
import { renderMedia, selectComposition } from "@remotion/renderer";
|
||||
import { imagesReelParamsSchema, type ImagesReelResult } from "@shared/reel";
|
||||
import { ffmpegService } from "../ffmpeg";
|
||||
import { ffmpegService, type TimedWord } from "../ffmpeg";
|
||||
import { generateVideoThumbnail } from "../thumbnail";
|
||||
import type { JobContext } from "./queue";
|
||||
import { resolveGeminiApiKey, resolveLogoPath } from "./assets";
|
||||
@@ -88,43 +87,31 @@ export function stripForTTS(text: string): string {
|
||||
.trim();
|
||||
}
|
||||
|
||||
/** Nombre de syllabes d'un mot français (groupes de voyelles). */
|
||||
function countSyllablesFr(word: string): number {
|
||||
const clean = word.replace(/[^a-zàâéèêëîïôùûüç]/gi, "").toLowerCase();
|
||||
if (!clean) return 1;
|
||||
return Math.max(1, clean.match(/[aeiouyàâéèêëîïôùûü]+/gi)?.length ?? 1);
|
||||
const EDGE_PUNCT = /^[.,!?;:…«»"'()\[\]]+|[.,!?;:…«»"'()\[\]]+$/g;
|
||||
|
||||
export interface WordTiming {
|
||||
word: string;
|
||||
startFrame: number;
|
||||
endFrame: number;
|
||||
}
|
||||
|
||||
const PUNCT_ONLY = /^[.,!?;:…\-—«»"''()\[\]]+$/;
|
||||
const EDGE_PUNCT = /^[.,!?;:…«»"''()\[\]]+|[.,!?;:…«»"''()\[\]]+$/g;
|
||||
|
||||
/**
|
||||
* Timings des mots prononcés (en images), répartis au prorata des syllabes.
|
||||
* Estimation provisoire : remplacée par les vrais timings de la voix au lot
|
||||
* « voix et sous-titres ».
|
||||
* Convertit le minutage réel des mots (en secondes, fourni par la voix) en
|
||||
* images. Chaque mot reste affiché jusqu'au début du suivant : pas de trou
|
||||
* pendant les respirations.
|
||||
*/
|
||||
export function computeWordTimings(
|
||||
displayText: string,
|
||||
audioDurationSeconds: number,
|
||||
fps: number,
|
||||
startFrame: number,
|
||||
): Array<{ word: string; startFrame: number; endFrame: number }> {
|
||||
const spokenWords = stripForTTS(displayText)
|
||||
.split(/\s+/)
|
||||
.filter((w) => w && !PUNCT_ONLY.test(w));
|
||||
if (spokenWords.length === 0) return [];
|
||||
|
||||
const cleanWords = spokenWords.map((w) => w.replace(EDGE_PUNCT, "") || w);
|
||||
const syllables = cleanWords.map(countSyllablesFr);
|
||||
const totalSyllables = syllables.reduce((a, b) => a + b, 0);
|
||||
const totalFrames = audioDurationSeconds * fps;
|
||||
|
||||
let currentFrame = startFrame;
|
||||
return cleanWords.map((word, i) => {
|
||||
const wordStart = currentFrame;
|
||||
currentFrame += Math.round((syllables[i] / totalSyllables) * totalFrames);
|
||||
return { word, startFrame: wordStart, endFrame: currentFrame };
|
||||
});
|
||||
export function toWordTimings(words: TimedWord[], fps: number): WordTiming[] {
|
||||
const timings = words
|
||||
.map((w) => ({ word: w.text.replace(EDGE_PUNCT, "") || w.text, start: w.start, end: w.end }))
|
||||
.filter((w) => w.word.trim());
|
||||
return timings.map((w, i) => ({
|
||||
word: w.word,
|
||||
startFrame: Math.round(w.start * fps),
|
||||
endFrame: Math.max(
|
||||
Math.round(w.start * fps) + 1,
|
||||
Math.round((i + 1 < timings.length ? timings[i + 1].start : w.end) * fps),
|
||||
),
|
||||
}));
|
||||
}
|
||||
|
||||
export async function runImagesReelJob({ job, progress }: JobContext): Promise<ImagesReelResult> {
|
||||
@@ -141,32 +128,27 @@ export async function runImagesReelJob({ job, progress }: JobContext): Promise<I
|
||||
|
||||
// --- Voix ---
|
||||
let audioUrl: string | undefined;
|
||||
let wordTimings: ReturnType<typeof computeWordTimings> | undefined;
|
||||
let wordTimings: WordTiming[] | undefined;
|
||||
let audioDuration = 0;
|
||||
|
||||
const ttsText = overlayText ? stripForTTS(overlayText) : "";
|
||||
if (overlayText && ttsText) {
|
||||
if (params.ttsEnabled && overlayText && ttsText) {
|
||||
await progress(15, "voice");
|
||||
const geminiApiKey = await resolveGeminiApiKey(params.ttsEngine);
|
||||
const tts = await ffmpegService.previewTTS(ttsText, params.ttsVoice, params.ttsEngine, geminiApiKey);
|
||||
if (!tts.success || !tts.audioBase64) {
|
||||
throw new Error(`La voix n'a pas pu être générée : ${tts.error ?? "réponse vide"}`);
|
||||
}
|
||||
const voice = await ffmpegService.previewVoice(ttsText, {
|
||||
voice: params.ttsVoice,
|
||||
engine: params.ttsEngine,
|
||||
style: params.ttsStyle,
|
||||
geminiApiKey: await resolveGeminiApiKey(params.ttsEngine),
|
||||
});
|
||||
for (const warning of voice.warnings) console.warn(`⚠️ [Reels] ${warning}`);
|
||||
|
||||
const audioFilename = `tts-${job.id}.mp3`;
|
||||
const audioPath = path.join(REMOTION_TEMP_DIR, audioFilename);
|
||||
const audioBuffer = Buffer.from(tts.audioBase64, "base64");
|
||||
await fs.promises.writeFile(audioPath, audioBuffer);
|
||||
await fs.promises.writeFile(audioPath, voice.audio);
|
||||
jobTempFiles.push(audioPath);
|
||||
audioUrl = localHttpUrl(`/uploads/temp/${audioFilename}`);
|
||||
|
||||
try {
|
||||
audioDuration = (await musicMetadata.parseFile(audioPath)).format.duration ?? 0;
|
||||
} catch {
|
||||
audioDuration = audioBuffer.length / 16000; // estimation à 128 kb/s
|
||||
}
|
||||
audioDuration = Math.max(audioDuration, ttsText.split(/\s+/).length * 0.35);
|
||||
wordTimings = computeWordTimings(overlayText, audioDuration, FPS, 0);
|
||||
audioDuration = voice.duration;
|
||||
wordTimings = toWordTimings(voice.words, FPS);
|
||||
}
|
||||
|
||||
// --- Durée : 25 à 30 s ---
|
||||
|
||||
@@ -94,6 +94,7 @@ export async function startReelWorker(): Promise<void> {
|
||||
console.error("❌ [ReelQueue] Récupération des jobs orphelins impossible :", error);
|
||||
}
|
||||
|
||||
console.log(`🎬 [ReelQueue] Worker démarré (traitements : ${Array.from(handlers.keys()).join(", ")})`);
|
||||
pollTimer = setInterval(kick, POLL_INTERVAL_MS);
|
||||
pollTimer.unref();
|
||||
kick();
|
||||
|
||||
@@ -30,14 +30,14 @@ export async function runVideoReelJob({ job, progress }: JobContext) {
|
||||
|
||||
await progress(15, "render");
|
||||
const startedAt = Date.now();
|
||||
const rendered = await ffmpegService.processReelFromUrl(resolveInternalUrl(media.originalUrl), {
|
||||
const rendered = await ffmpegService.renderReel(resolveInternalUrl(media.originalUrl), {
|
||||
text: params.overlayText,
|
||||
musicUrl,
|
||||
ttsEnabled: params.ttsEnabled,
|
||||
ttsVoice: params.ttsVoice,
|
||||
ttsEngine: params.ttsEngine,
|
||||
ttsStyle: params.ttsStyle,
|
||||
geminiApiKey,
|
||||
wordDuration: params.wordDuration,
|
||||
fontSize: params.fontSize,
|
||||
musicVolume: params.musicVolume,
|
||||
drawText: params.drawText,
|
||||
@@ -46,18 +46,10 @@ export async function runVideoReelJob({ job, progress }: JobContext) {
|
||||
storeName: params.storeName,
|
||||
enableEndingEffect: params.enableEndingEffect,
|
||||
});
|
||||
console.log(`⏱️ [Reels] FFmpeg : ${((Date.now() - startedAt) / 1000).toFixed(1)} s`);
|
||||
|
||||
if (!rendered.success || !rendered.videoBase64) {
|
||||
throw new Error(rendered.error || "Erreur de traitement vidéo FFmpeg");
|
||||
}
|
||||
if (rendered.ttsError) {
|
||||
// Voix demandée mais absente : ne jamais publier un Reel muet sans le dire
|
||||
throw new Error(`La voix n'a pas pu être générée : ${rendered.ttsError}`);
|
||||
}
|
||||
console.log(`⏱️ [Reels] Rendu : ${((Date.now() - startedAt) / 1000).toFixed(1)} s, vidéo de ${rendered.duration.toFixed(1)} s`);
|
||||
|
||||
await progress(65, "store");
|
||||
const videoBuffer = Buffer.from(rendered.videoBase64, "base64");
|
||||
const videoBuffer = rendered.video;
|
||||
const processedMedia = await storeRenderedVideo(job.userId, videoBuffer, `reel-${Date.now()}.mp4`);
|
||||
await storage.updatePostMedia(postId, [processedMedia.id]);
|
||||
|
||||
|
||||
+35
-62
@@ -1,5 +1,15 @@
|
||||
import { ffmpegService } from './ffmpeg';
|
||||
import * as musicMetadata from 'music-metadata';
|
||||
/**
|
||||
* Estimation de la durée de lecture d'un texte, affichée pendant la saisie.
|
||||
*
|
||||
* Auparavant, chaque pause de frappe générait une voix complète pour la
|
||||
* mesurer (un appel Gemini facturé à chaque fois). Le minutage réel des mots
|
||||
* vient désormais de la voix au moment du rendu ; ici, une estimation suffit.
|
||||
*/
|
||||
|
||||
/** Débit moyen d'une voix de synthèse française, en mots par seconde. */
|
||||
const WORDS_PER_SECOND = 2.6;
|
||||
/** Au-delà, la voix dépasse la durée confortable d'un Reel. */
|
||||
const MAX_COMFORTABLE_SECONDS = 45;
|
||||
|
||||
export interface SyncTiming {
|
||||
wordDuration: number;
|
||||
@@ -10,67 +20,30 @@ export interface SyncTiming {
|
||||
warnings: string[];
|
||||
}
|
||||
|
||||
export class TtsSyncService {
|
||||
/**
|
||||
* Calcule le word_duration optimal pour synchroniser l'affichage du texte
|
||||
* avec la durée réelle de la voix TTS générée.
|
||||
*/
|
||||
async calculateSyncTiming(
|
||||
text: string,
|
||||
voice: string,
|
||||
ttsEngine?: string,
|
||||
geminiApiKey?: string
|
||||
): Promise<SyncTiming> {
|
||||
const cleanText = this.cleanText(text);
|
||||
export function estimateVoiceTiming(text: string): SyncTiming {
|
||||
const words = text
|
||||
.replace(/#[\wÀ-ÿ]+/g, '')
|
||||
.replace(/https?:\/\/\S+/g, '')
|
||||
.split(/\s+/)
|
||||
.filter((w) => /[A-Za-z0-9À-ÿ]/.test(w));
|
||||
const pauses = (text.match(/[.!?;:]/g) ?? []).length;
|
||||
const punctuationPause = 0.35;
|
||||
const audioDuration = words.length / WORDS_PER_SECOND + pauses * punctuationPause;
|
||||
|
||||
// 1. Générer le TTS preview (avec la même voix que le rendu) et mesurer sa durée exacte
|
||||
const ttsResult = await ffmpegService.previewTTS(cleanText, voice, ttsEngine, geminiApiKey);
|
||||
if (!ttsResult.success || !ttsResult.audioBase64) {
|
||||
throw new Error('TTS preview failed: ' + (ttsResult.error || 'unknown'));
|
||||
}
|
||||
|
||||
const audioBuffer = Buffer.from(ttsResult.audioBase64, 'base64');
|
||||
const metadata = await musicMetadata.parseBuffer(audioBuffer, 'audio/mpeg');
|
||||
const audioDuration = metadata.format.duration || 0;
|
||||
|
||||
// 2. Analyser le texte (compte les mots réellement lus par la voix)
|
||||
const wordCount = this.calculateWordCount(cleanText);
|
||||
|
||||
// 3. Calculer le word_duration
|
||||
const wordDuration = wordCount > 0 ? audioDuration / wordCount : 0.6;
|
||||
|
||||
// 4. Validation
|
||||
const warnings: string[] = [];
|
||||
const isHealthy = wordDuration >= 0.25 && wordDuration <= 1.5;
|
||||
if (wordDuration < 0.25) {
|
||||
warnings.push('Texte trop long : les mots défileront très vite. Envisagez de raccourcir.');
|
||||
}
|
||||
if (wordDuration > 1.2) {
|
||||
warnings.push('Texte très court : affichage lent.');
|
||||
}
|
||||
|
||||
return {
|
||||
wordDuration,
|
||||
audioDuration,
|
||||
wordCount,
|
||||
punctuationPause: 0,
|
||||
isHealthy,
|
||||
warnings,
|
||||
};
|
||||
const warnings: string[] = [];
|
||||
if (audioDuration > MAX_COMFORTABLE_SECONDS) {
|
||||
warnings.push(`Texte long : environ ${Math.round(audioDuration)} s de voix. Visez moins de ${MAX_COMFORTABLE_SECONDS} s.`);
|
||||
}
|
||||
if (words.length > 0 && words.length < 4) {
|
||||
warnings.push('Texte très court : la voix ne durera que quelques secondes.');
|
||||
}
|
||||
|
||||
private cleanText(text: string): string {
|
||||
return text
|
||||
.replace(/#\w+/g, '')
|
||||
.replace(/https?:\/\/\S+/g, '')
|
||||
.replace(/[\uD83C-\uD83E][\uDC00-\uDFFF]|[☀-⛿✀-➿]/g, '')
|
||||
.trim();
|
||||
}
|
||||
|
||||
private calculateWordCount(text: string): number {
|
||||
const tokens = text.split(/\s+/).filter(w => w.length > 0);
|
||||
return tokens.filter(w => /[a-zA-Z0-9À-ſ]/.test(w)).length;
|
||||
}
|
||||
return {
|
||||
wordDuration: words.length ? audioDuration / words.length : 0,
|
||||
audioDuration,
|
||||
wordCount: words.length,
|
||||
punctuationPause,
|
||||
isHealthy: warnings.length === 0,
|
||||
warnings,
|
||||
};
|
||||
}
|
||||
|
||||
export const ttsSyncService = new TtsSyncService();
|
||||
+16
-3
@@ -5,6 +5,7 @@
|
||||
*/
|
||||
|
||||
import { z } from "zod";
|
||||
import { TTS_ENGINES, TTS_STYLES } from "./voices";
|
||||
|
||||
const optionalText = z
|
||||
.string()
|
||||
@@ -24,9 +25,11 @@ export const videoReelParamsSchema = z.object({
|
||||
description: optionalText,
|
||||
ttsEnabled: z.boolean().default(false),
|
||||
ttsVoice: optionalText,
|
||||
ttsEngine: z.enum(["gemini", "edge"]).optional(),
|
||||
ttsEngine: z.enum(TTS_ENGINES).optional(),
|
||||
ttsStyle: z.enum(TTS_STYLES).optional(),
|
||||
scheduledFor: optionalText,
|
||||
wordDuration: z.number().positive().max(5).default(0.6),
|
||||
// Ancien réglage, ignoré : le minutage vient désormais de la voix elle-même
|
||||
wordDuration: z.number().optional(),
|
||||
fontSize: z.number().int().min(16).max(200).default(64),
|
||||
musicVolume: z.number().min(0).max(2).default(0.25),
|
||||
drawText: z.boolean().default(true),
|
||||
@@ -45,13 +48,23 @@ export const imagesReelParamsSchema = z.object({
|
||||
overlayText: optionalText,
|
||||
musicUrl: optionalText,
|
||||
musicVolume: z.number().min(0).max(2).default(0.3),
|
||||
ttsEngine: z.enum(["gemini", "edge"]).optional(),
|
||||
ttsEnabled: z.boolean().default(true),
|
||||
ttsEngine: z.enum(TTS_ENGINES).optional(),
|
||||
ttsVoice: optionalText,
|
||||
ttsStyle: z.enum(TTS_STYLES).optional(),
|
||||
storeName: z.string().optional(),
|
||||
// Fichiers temporaires à supprimer une fois le rendu terminé
|
||||
tempFiles: z.array(z.string()).default([]),
|
||||
});
|
||||
|
||||
/** Aperçu de la voix. */
|
||||
export const ttsPreviewSchema = z.object({
|
||||
text: z.string({ required_error: "Texte requis" }).trim().min(1, "Texte requis").max(2000),
|
||||
ttsVoice: optionalText,
|
||||
ttsEngine: z.enum(TTS_ENGINES).optional(),
|
||||
ttsStyle: z.enum(TTS_STYLES).optional(),
|
||||
});
|
||||
|
||||
export type ImagesReelParams = z.infer<typeof imagesReelParamsSchema>;
|
||||
|
||||
export type ReelJobKind = "video" | "images";
|
||||
|
||||
@@ -0,0 +1,76 @@
|
||||
/**
|
||||
* Voix et styles de lecture proposés pour les Reels.
|
||||
* Même catalogue que le service Python (ffmpeg-service/app/voices.py).
|
||||
*/
|
||||
|
||||
export const TTS_ENGINES = ["gemini", "edge"] as const;
|
||||
export type TtsEngine = (typeof TTS_ENGINES)[number];
|
||||
|
||||
export const TTS_STYLES = ["neutral", "dynamic", "warm", "calm", "promo"] as const;
|
||||
export type TtsStyle = (typeof TTS_STYLES)[number];
|
||||
|
||||
export interface VoiceOption {
|
||||
id: string;
|
||||
label: string;
|
||||
gender: "female" | "male";
|
||||
}
|
||||
|
||||
export const TTS_STYLE_OPTIONS: { id: TtsStyle; label: string; description: string }[] = [
|
||||
{ id: "dynamic", label: "Dynamique", description: "Enthousiaste, rythme entraînant" },
|
||||
{ id: "warm", label: "Chaleureux", description: "Souriant et proche" },
|
||||
{ id: "promo", label: "Promo", description: "Annonce énergique des offres" },
|
||||
{ id: "calm", label: "Calme", description: "Posé et rassurant" },
|
||||
{ id: "neutral", label: "Neutre", description: "Lecture sans consigne" },
|
||||
];
|
||||
|
||||
export const GEMINI_VOICES: VoiceOption[] = [
|
||||
{ id: "Kore", label: "Kore — ferme", gender: "female" },
|
||||
{ id: "Aoede", label: "Aoede — légère", gender: "female" },
|
||||
{ id: "Leda", label: "Leda — jeune", gender: "female" },
|
||||
{ id: "Zephyr", label: "Zephyr — lumineuse", gender: "female" },
|
||||
{ id: "Callirrhoe", label: "Callirrhoe — décontractée", gender: "female" },
|
||||
{ id: "Autonoe", label: "Autonoe — lumineuse", gender: "female" },
|
||||
{ id: "Despina", label: "Despina — douce", gender: "female" },
|
||||
{ id: "Erinome", label: "Erinome — claire", gender: "female" },
|
||||
{ id: "Laomedeia", label: "Laomedeia — enjouée", gender: "female" },
|
||||
{ id: "Achernar", label: "Achernar — tendre", gender: "female" },
|
||||
{ id: "Gacrux", label: "Gacrux — mûre", gender: "female" },
|
||||
{ id: "Pulcherrima", label: "Pulcherrima — affirmée", gender: "female" },
|
||||
{ id: "Vindemiatrix", label: "Vindemiatrix — délicate", gender: "female" },
|
||||
{ id: "Sulafat", label: "Sulafat — chaleureuse", gender: "female" },
|
||||
{ id: "Charon", label: "Charon — informative", gender: "male" },
|
||||
{ id: "Puck", label: "Puck — enjouée", gender: "male" },
|
||||
{ id: "Fenrir", label: "Fenrir — enthousiaste", gender: "male" },
|
||||
{ id: "Orus", label: "Orus — ferme", gender: "male" },
|
||||
{ id: "Enceladus", label: "Enceladus — soufflée", gender: "male" },
|
||||
{ id: "Iapetus", label: "Iapetus — claire", gender: "male" },
|
||||
{ id: "Umbriel", label: "Umbriel — décontractée", gender: "male" },
|
||||
{ id: "Algieba", label: "Algieba — veloutée", gender: "male" },
|
||||
{ id: "Algenib", label: "Algenib — rocailleuse", gender: "male" },
|
||||
{ id: "Rasalgethi", label: "Rasalgethi — informative", gender: "male" },
|
||||
{ id: "Alnilam", label: "Alnilam — ferme", gender: "male" },
|
||||
{ id: "Schedar", label: "Schedar — posée", gender: "male" },
|
||||
{ id: "Achird", label: "Achird — amicale", gender: "male" },
|
||||
{ id: "Zubenelgenubi", label: "Zubenelgenubi — naturelle", gender: "male" },
|
||||
{ id: "Sadachbia", label: "Sadachbia — vive", gender: "male" },
|
||||
{ id: "Sadaltager", label: "Sadaltager — experte", gender: "male" },
|
||||
];
|
||||
|
||||
export const EDGE_VOICES: VoiceOption[] = [
|
||||
{ id: "fr-FR-VivienneMultilingualNeural", label: "Vivienne", gender: "female" },
|
||||
{ id: "fr-FR-DeniseNeural", label: "Denise", gender: "female" },
|
||||
{ id: "fr-FR-EloiseNeural", label: "Eloise", gender: "female" },
|
||||
{ id: "fr-FR-RemyMultilingualNeural", label: "Rémy", gender: "male" },
|
||||
{ id: "fr-FR-HenriNeural", label: "Henri", gender: "male" },
|
||||
];
|
||||
|
||||
export const DEFAULT_VOICE: Record<TtsEngine, string> = {
|
||||
gemini: "Kore",
|
||||
edge: "fr-FR-VivienneMultilingualNeural",
|
||||
};
|
||||
|
||||
export const DEFAULT_TTS_STYLE: TtsStyle = "dynamic";
|
||||
|
||||
export function voicesFor(engine: TtsEngine): VoiceOption[] {
|
||||
return engine === "gemini" ? GEMINI_VOICES : EDGE_VOICES;
|
||||
}
|
||||
Reference in new issue
Block a user