add TTS voice

This commit is contained in:
Michael committed 2026-01-22 15:28:00 +01:00
1 parent 44928c61b8
commit 9ffd92f6bf
5 files changed
+214 -15

No files matched your search

+62 -1
View File
@@ -4,7 +4,7 @@ import { useLocation } from "wouter";
import {
Send, Sparkles, Video, Music, Type, Calendar,
Upload, Camera, Play, Pause, Volume2, VolumeX,
ChevronRight, Loader2, Check, RefreshCw
ChevronRight, Loader2, Check, RefreshCw, Mic
} from "lucide-react";
import { useDropzone } from "react-dropzone";
import Sidebar from "@/components/sidebar";
@@ -14,8 +14,18 @@ import { Card, CardContent, CardDescription, CardHeader, CardTitle } from "@/com
import { Textarea } from "@/components/ui/textarea";
import { Input } from "@/components/ui/input";
import { Label } from "@/components/ui/label";
import { Input } from "@/components/ui/input";
import { Label } from "@/components/ui/label";
import { Checkbox } from "@/components/ui/checkbox";
import { Switch } from "@/components/ui/switch";
import { Slider } from "@/components/ui/slider";
import {
Select,
SelectContent,
SelectItem,
SelectTrigger,
SelectValue,
} from "@/components/ui/select";
import { useToast } from "@/hooks/use-toast";
import { apiRequest, queryClient } from "@/lib/queryClient";
import type { SocialPage, Media } from "@shared/schema";
@@ -55,7 +65,10 @@ export default function NewReel() {
const [generatedVariants, setGeneratedVariants] = useState<any[]>([]);
const [selectedPages, setSelectedPages] = useState<string[]>([]);
const [scheduledDate, setScheduledDate] = useState<Date | undefined>(undefined);
const [scheduledDate, setScheduledDate] = useState<Date | undefined>(undefined);
const [musicVolume, setMusicVolume] = useState([25]);
const [ttsEnabled, setTtsEnabled] = useState(false);
const [ttsVoice, setTtsVoice] = useState("female");
// État audio preview
const [isPlaying, setIsPlaying] = useState<string | null>(null);
@@ -298,6 +311,8 @@ export default function NewReel() {
pageIds: selectedPages,
scheduledFor: scheduledDate?.toISOString(),
musicVolume: musicVolume[0] / 100,
ttsEnabled,
ttsVoice,
});
};
@@ -685,6 +700,52 @@ export default function NewReel() {
rows={4}
/>
<div className="flex items-center space-x-2 mt-4">
<Switch
id="tts-mode"
checked={ttsEnabled}
onCheckedChange={setTtsEnabled}
/>
<Label htmlFor="tts-mode" className="font-medium cursor-pointer">
Activer la lecture voix (TTS)
</Label>
</div>
{ttsEnabled && (
<div className="mt-4 space-y-2 ml-12 p-4 bg-muted/30 rounded-lg border border-border/50">
<Label className="flex items-center gap-2">
<Mic className="w-4 h-4" />
Voix du narrateur
</Label>
<div className="grid grid-cols-2 gap-4">
<div
onClick={() => setTtsVoice("female")}
className={`cursor-pointer p-3 rounded-md border-2 text-center transition-all ${ttsVoice === 'female'
? 'border-primary bg-primary/10'
: 'border-transparent bg-background hover:bg-accent'
}`}
>
<div className="font-semibold">Féminine</div>
<div className="text-xs text-muted-foreground">Vivienne</div>
</div>
<div
onClick={() => setTtsVoice("male")}
className={`cursor-pointer p-3 rounded-md border-2 text-center transition-all ${ttsVoice === 'male'
? 'border-primary bg-primary/10'
: 'border-transparent bg-background hover:bg-accent'
}`}
>
<div className="font-semibold">Masculine</div>
<div className="text-xs text-muted-foreground">Rémy</div>
</div>
</div>
<p className="text-xs text-muted-foreground mt-2">
Le texte sera automatiquement synchronisé avec la voix.
Les #hashtags et émojis ne seront pas lus.
</p>
</div>
)}
<div className="mt-4 flex justify-between">
<Button variant="outline" onClick={() => setCurrentStep('music')}>
Retour
+124 -14
View File
@@ -9,6 +9,9 @@ import base64
import requests
import shutil
from pathlib import Path
import edge_tts
import re
import emoji
app = FastAPI()
@@ -29,6 +32,34 @@ class ReelRequest(BaseModel):
word_duration: float = 0.6
font_size: int = 60
music_volume: float = 0.25
tts_enabled: bool = False
tts_voice: str = "fr-FR-VivienneNeural"
def clean_text_for_tts(text: str) -> str:
# 1. Remove emojis
text = emoji.replace_emoji(text, replace="")
# 2. Remove hashtags (e.g. #viral #reels)
text = re.sub(r"#\w+", "", text)
# 3. Cleanup whitespace
return " ".join(text.split())
async def generate_tts_with_subs(
text: str, voice: str, audio_path: Path, vtt_path: Path
):
communicate = edge_tts.Communicate(text, voice)
submaker = edge_tts.SubMaker()
with open(audio_path, "wb") as file:
async for chunk in communicate.stream():
if chunk["type"] == "audio":
file.write(chunk["data"])
elif chunk["type"] == "WordBoundary":
submaker.feed(chunk)
with open(vtt_path, "w", encoding="utf-8") as file:
file.write(submaker.generate_subs())
class ReelResponse(BaseModel):
@@ -57,6 +88,8 @@ async def process_reel(request: ReelRequest, x_api_key: str = Header(None)):
input_video_path = job_dir / "input.mp4"
input_audio_path = job_dir / "music.mp3"
tts_audio_path = job_dir / "tts.mp3"
tts_vtt_path = job_dir / "tts.vtt"
output_video_path = job_dir / "output.mp4"
# 1. Save Input Video
@@ -86,25 +119,71 @@ async def process_reel(request: ReelRequest, x_api_key: str = Header(None)):
print(f"Failed to download music: {e}")
# We continue without music if it fails
# 3. Build FFmpeg Command
# 3. Generate TTS (if enabled)
has_tts = False
tts_clean_text = ""
if request.tts_enabled and request.text:
try:
# Clean text for TTS (remove hashtags/emojis)
tts_clean_text = clean_text_for_tts(request.text)
# Check for male/female voice map
voice = request.tts_voice
if voice == "male":
voice = "fr-FR-RemyNeural"
elif voice == "female":
voice = "fr-FR-VivienneNeural"
elif not voice or "Neural" not in voice:
# Default if invalid
voice = "fr-FR-VivienneNeural"
if tts_clean_text:
await generate_tts_with_subs(
tts_clean_text, voice, tts_audio_path, tts_vtt_path
)
has_tts = True
except Exception as e:
print(f"Failed to generate TTS: {e}")
# 4. Build FFmpeg Command
cmd = ["ffmpeg", "-y", "-i", str(input_video_path)]
video_filters = []
audio_filters = []
# Text Overlay (Video Filter)
# Text Overlay
# If TTS is enabled, we use the generated VTT subtitles for perfect sync
# If not, we use the standard drawtext
if request.text:
# Escape text for drawtext
sanitized_text = request.text.replace("'", "").replace(":", "\\:")
# Move text to bottom (with 150px padding to avoid UI elements)
drawtext = f"drawtext=fontfile={FONT_PATH}:text='{sanitized_text}':fontcolor=white:fontsize={request.font_size}:box=1:boxcolor=black@0.5:boxborderw=5:x=(w-text_w)/2:y=h-text_h-150"
video_filters.append(drawtext)
if has_tts:
# Use subtitles filter
# Force style to look like TikTok/Reels text (Bottom center, white, black box)
style = f"FontName=Arial,FontSize={request.font_size},PrimaryColour=&H00FFFFFF,OutlineColour=&H80000000,BorderStyle=3,BackColour=&H80000000,Bold=1,Alignment=2,MarginV=150"
# Escape path for FFmpeg filter
vtt_path_str = str(tts_vtt_path).replace("\\", "/").replace(":", "\\:")
video_filters.append(
f"subtitles='{vtt_path_str}':force_style='{style}'"
)
else:
# Standard Drawtext logic
sanitized_text = request.text.replace("'", "").replace(":", "\\:")
drawtext = f"drawtext=fontfile={FONT_PATH}:text='{sanitized_text}':fontcolor=white:fontsize={request.font_size}:box=1:boxcolor=black@0.5:boxborderw=5:x=(w-text_w)/2:y=h-text_h-150"
video_filters.append(drawtext)
# Audio Mixing/Volume (Audio Filter)
if has_music:
cmd.extend(["-i", str(input_audio_path)])
# Apply volume adjustment to the music
audio_filters.append(f"volume={request.music_volume}")
# If TTS is present, reduce music volume further to prioritize voice
music_vol = request.music_volume * 0.5 if has_tts else request.music_volume
audio_filters.append(f"[1:a]volume={music_vol}[music]")
if has_tts:
cmd.extend(["-i", str(tts_audio_path)])
# TTS is typically input 2 if music exists, or input 1 if no music
tts_input_idx = 2 if has_music else 1
audio_filters.append(f"[{tts_input_idx}:a]volume=1.5[voice]")
# Apply Video Filters if any
if video_filters:
@@ -112,14 +191,45 @@ async def process_reel(request: ReelRequest, x_api_key: str = Header(None)):
# Apply Audio Filters if any
# Note: -af applies to the output audio stream.
# Apply Audio Filters if any
if audio_filters:
cmd.extend(["-af", ",".join(audio_filters)])
# If we have multiple audio sources, we need to mix them
if has_music and has_tts:
# Mix music and voice
filter_complex = (
";".join(audio_filters)
+ ";[music][voice]amix=inputs=2:duration=longest[aout]"
)
cmd.extend(["-filter_complex", filter_complex])
# We need complex filter for mixing, so we don't use -vf/-af separately for audio
# But we still need video filters
if video_filters:
# Remove the previously added -vf and use filter_complex for everything ideally,
# or just keep -vf for video simple chain if separate.
# FFmpeg allows -vf and -filter_complex together if they touch different streams.
pass
elif has_music:
# Valid because we modified the filter previously to verify [1:a]... which requires filter_complex or mapping
# Let's simplify: if simple volume filter, use -af. If named pads ([music]), use complex.
# To keep it robust, let's use filter_complex for audio always if we started naming pads.
cmd.extend(
["-filter_complex", f"[1:a]volume={request.music_volume}[aout]"]
)
elif has_tts:
cmd.extend(["-filter_complex", f"[1:a]volume=1.5[aout]"])
if has_music:
# Map video from input 0
# Map audio from input 1 (music)
# -shortest: finish when the shortest input (usually video or music) ends
cmd.extend(["-map", "0:v", "-map", "1:a", "-shortest"])
# Note: The above logic replaces the simple -af append. We need to be careful not to double add.
# Let's Refactor slightly to ensure clean command construction.
if has_music or has_tts:
# Map processed video
cmd.extend(["-map", "0:v"])
# Map processed audio [aout]
cmd.extend(["-map", "[aout]"])
# -shortest: finish when the shortest input ends
cmd.extend(["-shortest"])
else:
# Keep original video and audio (if exists)
cmd.extend(["-map", "0:v", "-map", "0:a?"])
+9
View File
@@ -182,6 +182,8 @@ reelsRouter.post('/reels/preview', async (req: Request, res: Response) => {
musicTrackId,
musicUrl,
overlayText,
ttsEnabled,
ttsVoice,
wordDuration = 0.6,
fontSize = 60,
musicVolume = 0.25,
@@ -210,6 +212,8 @@ reelsRouter.post('/reels/preview', async (req: Request, res: Response) => {
const result = await ffmpegService.processReelFromUrl(media.originalUrl, {
text: overlayText,
musicUrl: finalMusicUrl,
ttsEnabled,
ttsVoice,
wordDuration,
fontSize,
musicVolume,
@@ -244,6 +248,8 @@ reelsRouter.post('/reels', async (req: Request, res: Response) => {
musicUrl,
overlayText,
description,
ttsEnabled,
ttsVoice,
pageIds,
scheduledFor,
wordDuration = 0.6,
@@ -283,6 +289,7 @@ reelsRouter.post('/reels', async (req: Request, res: Response) => {
videoMediaId,
hasMusic: !!finalMusicUrl,
hasText: !!overlayText,
hasTTS: !!ttsEnabled,
pageCount: pageIds.length,
});
@@ -290,6 +297,8 @@ reelsRouter.post('/reels', async (req: Request, res: Response) => {
const ffmpegResult = await ffmpegService.processReelFromUrl(media.originalUrl, {
text: overlayText,
musicUrl: finalMusicUrl,
ttsEnabled,
ttsVoice,
wordDuration,
fontSize,
musicVolume,
+12
View File
@@ -11,6 +11,8 @@ interface FFmpegReelRequest {
text?: string; // Texte overlay style TikTok
music_id?: string; // ID de la musique (catalogue FFmpeg)
music_url?: string; // OU URL directe de la musique
tts_enabled?: boolean; // Activation du TTS
tts_voice?: string; // Voix TTS (ex: fr-FR-VivienneNeural)
word_duration?: number; // Durée par mot (default: 0.6s)
font_size?: number; // Taille police (default: 60)
music_volume?: number; // Volume musique (default: 0.25)
@@ -62,6 +64,8 @@ export class FFmpegService {
text?: string;
musicId?: string;
musicUrl?: string;
ttsEnabled?: boolean;
ttsVoice?: string;
wordDuration?: number;
fontSize?: number;
musicVolume?: number;
@@ -74,6 +78,8 @@ export class FFmpegService {
text: options.text,
music_id: options.musicId,
music_url: options.musicUrl,
tts_enabled: options.ttsEnabled,
tts_voice: options.ttsVoice,
word_duration: options.wordDuration ?? 0.6,
font_size: options.fontSize ?? 60,
music_volume: options.musicVolume ?? 0.25,
@@ -91,6 +97,7 @@ export class FFmpegService {
hasText: !!options.text,
hasMusicId: !!options.musicId,
hasMusicUrl: !!options.musicUrl,
hasTTS: options.ttsEnabled,
});
try {
@@ -147,6 +154,8 @@ export class FFmpegService {
text?: string;
musicId?: string;
musicUrl?: string;
ttsEnabled?: boolean;
ttsVoice?: string;
wordDuration?: number;
fontSize?: number;
musicVolume?: number;
@@ -159,6 +168,8 @@ export class FFmpegService {
text: options.text,
music_id: options.musicId,
music_url: options.musicUrl,
tts_enabled: options.ttsEnabled,
tts_voice: options.ttsVoice,
word_duration: options.wordDuration ?? 0.6,
font_size: options.fontSize ?? 60,
music_volume: options.musicVolume ?? 0.25,
@@ -176,6 +187,7 @@ export class FFmpegService {
hasText: !!options.text,
hasMusicId: !!options.musicId,
hasMusicUrl: !!options.musicUrl,
hasTTS: options.ttsEnabled,
});
try {
+7
View File
@@ -45,6 +45,13 @@ Ajout d'une fonctionnalité complète de création de Reels Facebook permettant
- [x] Configurer FreeSound API (env vars)
- [x] Configurer FFmpeg API URL et clé (interface Settings)
- [x] Ajouter la voix off (TTS) gratuite via Edge-TTS
- [x] FFmpeg Service: Installer edge-tts et implémenter le mixage audio
- [x] Backend: Supporter les options TTS (enable, voice)
- [ ] Améliorer le TTS (Genre, Sync, Cleanup)
- [ ] FFmpeg: Nettoyer le texte (No hashtags/emojis) avant TTS
- [ ] FFmpeg: Générer fichier VTT pour synchro sous-titres
- [ ] Frontend: Sélecteur de voix (Homme/Femme)
- [/] Ajouter la valeur 'reel' à l'enum post_type en base de données
- [x] Amélioration UI Musique (Pagination, Preview Audio)