mirror of
https://github.com/R0m1k3/Socialflow.git
synced 2026-10-12 01:36:54 +02:00
add TTS voice
This commit is contained in:
1 parent
44928c61b8
commit
9ffd92f6bf
5 files changed
+214
-15
No files matched your search
@@ -4,7 +4,7 @@ import { useLocation } from "wouter";
|
||||
import {
|
||||
Send, Sparkles, Video, Music, Type, Calendar,
|
||||
Upload, Camera, Play, Pause, Volume2, VolumeX,
|
||||
ChevronRight, Loader2, Check, RefreshCw
|
||||
ChevronRight, Loader2, Check, RefreshCw, Mic
|
||||
} from "lucide-react";
|
||||
import { useDropzone } from "react-dropzone";
|
||||
import Sidebar from "@/components/sidebar";
|
||||
@@ -14,8 +14,18 @@ import { Card, CardContent, CardDescription, CardHeader, CardTitle } from "@/com
|
||||
import { Textarea } from "@/components/ui/textarea";
|
||||
import { Input } from "@/components/ui/input";
|
||||
import { Label } from "@/components/ui/label";
|
||||
import { Input } from "@/components/ui/input";
|
||||
import { Label } from "@/components/ui/label";
|
||||
import { Checkbox } from "@/components/ui/checkbox";
|
||||
import { Switch } from "@/components/ui/switch";
|
||||
import { Slider } from "@/components/ui/slider";
|
||||
import {
|
||||
Select,
|
||||
SelectContent,
|
||||
SelectItem,
|
||||
SelectTrigger,
|
||||
SelectValue,
|
||||
} from "@/components/ui/select";
|
||||
import { useToast } from "@/hooks/use-toast";
|
||||
import { apiRequest, queryClient } from "@/lib/queryClient";
|
||||
import type { SocialPage, Media } from "@shared/schema";
|
||||
@@ -55,7 +65,10 @@ export default function NewReel() {
|
||||
const [generatedVariants, setGeneratedVariants] = useState<any[]>([]);
|
||||
const [selectedPages, setSelectedPages] = useState<string[]>([]);
|
||||
const [scheduledDate, setScheduledDate] = useState<Date | undefined>(undefined);
|
||||
const [scheduledDate, setScheduledDate] = useState<Date | undefined>(undefined);
|
||||
const [musicVolume, setMusicVolume] = useState([25]);
|
||||
const [ttsEnabled, setTtsEnabled] = useState(false);
|
||||
const [ttsVoice, setTtsVoice] = useState("female");
|
||||
|
||||
// État audio preview
|
||||
const [isPlaying, setIsPlaying] = useState<string | null>(null);
|
||||
@@ -298,6 +311,8 @@ export default function NewReel() {
|
||||
pageIds: selectedPages,
|
||||
scheduledFor: scheduledDate?.toISOString(),
|
||||
musicVolume: musicVolume[0] / 100,
|
||||
ttsEnabled,
|
||||
ttsVoice,
|
||||
});
|
||||
};
|
||||
|
||||
@@ -685,6 +700,52 @@ export default function NewReel() {
|
||||
rows={4}
|
||||
/>
|
||||
|
||||
<div className="flex items-center space-x-2 mt-4">
|
||||
<Switch
|
||||
id="tts-mode"
|
||||
checked={ttsEnabled}
|
||||
onCheckedChange={setTtsEnabled}
|
||||
/>
|
||||
<Label htmlFor="tts-mode" className="font-medium cursor-pointer">
|
||||
Activer la lecture voix (TTS)
|
||||
</Label>
|
||||
</div>
|
||||
|
||||
{ttsEnabled && (
|
||||
<div className="mt-4 space-y-2 ml-12 p-4 bg-muted/30 rounded-lg border border-border/50">
|
||||
<Label className="flex items-center gap-2">
|
||||
<Mic className="w-4 h-4" />
|
||||
Voix du narrateur
|
||||
</Label>
|
||||
<div className="grid grid-cols-2 gap-4">
|
||||
<div
|
||||
onClick={() => setTtsVoice("female")}
|
||||
className={`cursor-pointer p-3 rounded-md border-2 text-center transition-all ${ttsVoice === 'female'
|
||||
? 'border-primary bg-primary/10'
|
||||
: 'border-transparent bg-background hover:bg-accent'
|
||||
}`}
|
||||
>
|
||||
<div className="font-semibold">Féminine</div>
|
||||
<div className="text-xs text-muted-foreground">Vivienne</div>
|
||||
</div>
|
||||
<div
|
||||
onClick={() => setTtsVoice("male")}
|
||||
className={`cursor-pointer p-3 rounded-md border-2 text-center transition-all ${ttsVoice === 'male'
|
||||
? 'border-primary bg-primary/10'
|
||||
: 'border-transparent bg-background hover:bg-accent'
|
||||
}`}
|
||||
>
|
||||
<div className="font-semibold">Masculine</div>
|
||||
<div className="text-xs text-muted-foreground">Rémy</div>
|
||||
</div>
|
||||
</div>
|
||||
<p className="text-xs text-muted-foreground mt-2">
|
||||
Le texte sera automatiquement synchronisé avec la voix.
|
||||
Les #hashtags et émojis ne seront pas lus.
|
||||
</p>
|
||||
</div>
|
||||
)}
|
||||
|
||||
<div className="mt-4 flex justify-between">
|
||||
<Button variant="outline" onClick={() => setCurrentStep('music')}>
|
||||
Retour
|
||||
|
||||
+124
-14
@@ -9,6 +9,9 @@ import base64
|
||||
import requests
|
||||
import shutil
|
||||
from pathlib import Path
|
||||
import edge_tts
|
||||
import re
|
||||
import emoji
|
||||
|
||||
app = FastAPI()
|
||||
|
||||
@@ -29,6 +32,34 @@ class ReelRequest(BaseModel):
|
||||
word_duration: float = 0.6
|
||||
font_size: int = 60
|
||||
music_volume: float = 0.25
|
||||
tts_enabled: bool = False
|
||||
tts_voice: str = "fr-FR-VivienneNeural"
|
||||
|
||||
|
||||
def clean_text_for_tts(text: str) -> str:
|
||||
# 1. Remove emojis
|
||||
text = emoji.replace_emoji(text, replace="")
|
||||
# 2. Remove hashtags (e.g. #viral #reels)
|
||||
text = re.sub(r"#\w+", "", text)
|
||||
# 3. Cleanup whitespace
|
||||
return " ".join(text.split())
|
||||
|
||||
|
||||
async def generate_tts_with_subs(
|
||||
text: str, voice: str, audio_path: Path, vtt_path: Path
|
||||
):
|
||||
communicate = edge_tts.Communicate(text, voice)
|
||||
submaker = edge_tts.SubMaker()
|
||||
|
||||
with open(audio_path, "wb") as file:
|
||||
async for chunk in communicate.stream():
|
||||
if chunk["type"] == "audio":
|
||||
file.write(chunk["data"])
|
||||
elif chunk["type"] == "WordBoundary":
|
||||
submaker.feed(chunk)
|
||||
|
||||
with open(vtt_path, "w", encoding="utf-8") as file:
|
||||
file.write(submaker.generate_subs())
|
||||
|
||||
|
||||
class ReelResponse(BaseModel):
|
||||
@@ -57,6 +88,8 @@ async def process_reel(request: ReelRequest, x_api_key: str = Header(None)):
|
||||
|
||||
input_video_path = job_dir / "input.mp4"
|
||||
input_audio_path = job_dir / "music.mp3"
|
||||
tts_audio_path = job_dir / "tts.mp3"
|
||||
tts_vtt_path = job_dir / "tts.vtt"
|
||||
output_video_path = job_dir / "output.mp4"
|
||||
|
||||
# 1. Save Input Video
|
||||
@@ -86,25 +119,71 @@ async def process_reel(request: ReelRequest, x_api_key: str = Header(None)):
|
||||
print(f"Failed to download music: {e}")
|
||||
# We continue without music if it fails
|
||||
|
||||
# 3. Build FFmpeg Command
|
||||
# 3. Generate TTS (if enabled)
|
||||
has_tts = False
|
||||
tts_clean_text = ""
|
||||
|
||||
if request.tts_enabled and request.text:
|
||||
try:
|
||||
# Clean text for TTS (remove hashtags/emojis)
|
||||
tts_clean_text = clean_text_for_tts(request.text)
|
||||
|
||||
# Check for male/female voice map
|
||||
voice = request.tts_voice
|
||||
if voice == "male":
|
||||
voice = "fr-FR-RemyNeural"
|
||||
elif voice == "female":
|
||||
voice = "fr-FR-VivienneNeural"
|
||||
elif not voice or "Neural" not in voice:
|
||||
# Default if invalid
|
||||
voice = "fr-FR-VivienneNeural"
|
||||
|
||||
if tts_clean_text:
|
||||
await generate_tts_with_subs(
|
||||
tts_clean_text, voice, tts_audio_path, tts_vtt_path
|
||||
)
|
||||
has_tts = True
|
||||
except Exception as e:
|
||||
print(f"Failed to generate TTS: {e}")
|
||||
|
||||
# 4. Build FFmpeg Command
|
||||
cmd = ["ffmpeg", "-y", "-i", str(input_video_path)]
|
||||
|
||||
video_filters = []
|
||||
audio_filters = []
|
||||
|
||||
# Text Overlay (Video Filter)
|
||||
# Text Overlay
|
||||
# If TTS is enabled, we use the generated VTT subtitles for perfect sync
|
||||
# If not, we use the standard drawtext
|
||||
if request.text:
|
||||
# Escape text for drawtext
|
||||
sanitized_text = request.text.replace("'", "").replace(":", "\\:")
|
||||
# Move text to bottom (with 150px padding to avoid UI elements)
|
||||
drawtext = f"drawtext=fontfile={FONT_PATH}:text='{sanitized_text}':fontcolor=white:fontsize={request.font_size}:box=1:boxcolor=black@0.5:boxborderw=5:x=(w-text_w)/2:y=h-text_h-150"
|
||||
video_filters.append(drawtext)
|
||||
if has_tts:
|
||||
# Use subtitles filter
|
||||
# Force style to look like TikTok/Reels text (Bottom center, white, black box)
|
||||
style = f"FontName=Arial,FontSize={request.font_size},PrimaryColour=&H00FFFFFF,OutlineColour=&H80000000,BorderStyle=3,BackColour=&H80000000,Bold=1,Alignment=2,MarginV=150"
|
||||
# Escape path for FFmpeg filter
|
||||
vtt_path_str = str(tts_vtt_path).replace("\\", "/").replace(":", "\\:")
|
||||
video_filters.append(
|
||||
f"subtitles='{vtt_path_str}':force_style='{style}'"
|
||||
)
|
||||
else:
|
||||
# Standard Drawtext logic
|
||||
sanitized_text = request.text.replace("'", "").replace(":", "\\:")
|
||||
drawtext = f"drawtext=fontfile={FONT_PATH}:text='{sanitized_text}':fontcolor=white:fontsize={request.font_size}:box=1:boxcolor=black@0.5:boxborderw=5:x=(w-text_w)/2:y=h-text_h-150"
|
||||
video_filters.append(drawtext)
|
||||
|
||||
# Audio Mixing/Volume (Audio Filter)
|
||||
if has_music:
|
||||
cmd.extend(["-i", str(input_audio_path)])
|
||||
# Apply volume adjustment to the music
|
||||
audio_filters.append(f"volume={request.music_volume}")
|
||||
# If TTS is present, reduce music volume further to prioritize voice
|
||||
music_vol = request.music_volume * 0.5 if has_tts else request.music_volume
|
||||
audio_filters.append(f"[1:a]volume={music_vol}[music]")
|
||||
|
||||
if has_tts:
|
||||
cmd.extend(["-i", str(tts_audio_path)])
|
||||
# TTS is typically input 2 if music exists, or input 1 if no music
|
||||
tts_input_idx = 2 if has_music else 1
|
||||
audio_filters.append(f"[{tts_input_idx}:a]volume=1.5[voice]")
|
||||
|
||||
# Apply Video Filters if any
|
||||
if video_filters:
|
||||
@@ -112,14 +191,45 @@ async def process_reel(request: ReelRequest, x_api_key: str = Header(None)):
|
||||
|
||||
# Apply Audio Filters if any
|
||||
# Note: -af applies to the output audio stream.
|
||||
# Apply Audio Filters if any
|
||||
if audio_filters:
|
||||
cmd.extend(["-af", ",".join(audio_filters)])
|
||||
# If we have multiple audio sources, we need to mix them
|
||||
if has_music and has_tts:
|
||||
# Mix music and voice
|
||||
filter_complex = (
|
||||
";".join(audio_filters)
|
||||
+ ";[music][voice]amix=inputs=2:duration=longest[aout]"
|
||||
)
|
||||
cmd.extend(["-filter_complex", filter_complex])
|
||||
# We need complex filter for mixing, so we don't use -vf/-af separately for audio
|
||||
# But we still need video filters
|
||||
if video_filters:
|
||||
# Remove the previously added -vf and use filter_complex for everything ideally,
|
||||
# or just keep -vf for video simple chain if separate.
|
||||
# FFmpeg allows -vf and -filter_complex together if they touch different streams.
|
||||
pass
|
||||
elif has_music:
|
||||
# Valid because we modified the filter previously to verify [1:a]... which requires filter_complex or mapping
|
||||
# Let's simplify: if simple volume filter, use -af. If named pads ([music]), use complex.
|
||||
# To keep it robust, let's use filter_complex for audio always if we started naming pads.
|
||||
cmd.extend(
|
||||
["-filter_complex", f"[1:a]volume={request.music_volume}[aout]"]
|
||||
)
|
||||
elif has_tts:
|
||||
cmd.extend(["-filter_complex", f"[1:a]volume=1.5[aout]"])
|
||||
|
||||
if has_music:
|
||||
# Map video from input 0
|
||||
# Map audio from input 1 (music)
|
||||
# -shortest: finish when the shortest input (usually video or music) ends
|
||||
cmd.extend(["-map", "0:v", "-map", "1:a", "-shortest"])
|
||||
# Note: The above logic replaces the simple -af append. We need to be careful not to double add.
|
||||
# Let's Refactor slightly to ensure clean command construction.
|
||||
|
||||
if has_music or has_tts:
|
||||
# Map processed video
|
||||
cmd.extend(["-map", "0:v"])
|
||||
|
||||
# Map processed audio [aout]
|
||||
cmd.extend(["-map", "[aout]"])
|
||||
|
||||
# -shortest: finish when the shortest input ends
|
||||
cmd.extend(["-shortest"])
|
||||
else:
|
||||
# Keep original video and audio (if exists)
|
||||
cmd.extend(["-map", "0:v", "-map", "0:a?"])
|
||||
|
||||
@@ -182,6 +182,8 @@ reelsRouter.post('/reels/preview', async (req: Request, res: Response) => {
|
||||
musicTrackId,
|
||||
musicUrl,
|
||||
overlayText,
|
||||
ttsEnabled,
|
||||
ttsVoice,
|
||||
wordDuration = 0.6,
|
||||
fontSize = 60,
|
||||
musicVolume = 0.25,
|
||||
@@ -210,6 +212,8 @@ reelsRouter.post('/reels/preview', async (req: Request, res: Response) => {
|
||||
const result = await ffmpegService.processReelFromUrl(media.originalUrl, {
|
||||
text: overlayText,
|
||||
musicUrl: finalMusicUrl,
|
||||
ttsEnabled,
|
||||
ttsVoice,
|
||||
wordDuration,
|
||||
fontSize,
|
||||
musicVolume,
|
||||
@@ -244,6 +248,8 @@ reelsRouter.post('/reels', async (req: Request, res: Response) => {
|
||||
musicUrl,
|
||||
overlayText,
|
||||
description,
|
||||
ttsEnabled,
|
||||
ttsVoice,
|
||||
pageIds,
|
||||
scheduledFor,
|
||||
wordDuration = 0.6,
|
||||
@@ -283,6 +289,7 @@ reelsRouter.post('/reels', async (req: Request, res: Response) => {
|
||||
videoMediaId,
|
||||
hasMusic: !!finalMusicUrl,
|
||||
hasText: !!overlayText,
|
||||
hasTTS: !!ttsEnabled,
|
||||
pageCount: pageIds.length,
|
||||
});
|
||||
|
||||
@@ -290,6 +297,8 @@ reelsRouter.post('/reels', async (req: Request, res: Response) => {
|
||||
const ffmpegResult = await ffmpegService.processReelFromUrl(media.originalUrl, {
|
||||
text: overlayText,
|
||||
musicUrl: finalMusicUrl,
|
||||
ttsEnabled,
|
||||
ttsVoice,
|
||||
wordDuration,
|
||||
fontSize,
|
||||
musicVolume,
|
||||
|
||||
@@ -11,6 +11,8 @@ interface FFmpegReelRequest {
|
||||
text?: string; // Texte overlay style TikTok
|
||||
music_id?: string; // ID de la musique (catalogue FFmpeg)
|
||||
music_url?: string; // OU URL directe de la musique
|
||||
tts_enabled?: boolean; // Activation du TTS
|
||||
tts_voice?: string; // Voix TTS (ex: fr-FR-VivienneNeural)
|
||||
word_duration?: number; // Durée par mot (default: 0.6s)
|
||||
font_size?: number; // Taille police (default: 60)
|
||||
music_volume?: number; // Volume musique (default: 0.25)
|
||||
@@ -62,6 +64,8 @@ export class FFmpegService {
|
||||
text?: string;
|
||||
musicId?: string;
|
||||
musicUrl?: string;
|
||||
ttsEnabled?: boolean;
|
||||
ttsVoice?: string;
|
||||
wordDuration?: number;
|
||||
fontSize?: number;
|
||||
musicVolume?: number;
|
||||
@@ -74,6 +78,8 @@ export class FFmpegService {
|
||||
text: options.text,
|
||||
music_id: options.musicId,
|
||||
music_url: options.musicUrl,
|
||||
tts_enabled: options.ttsEnabled,
|
||||
tts_voice: options.ttsVoice,
|
||||
word_duration: options.wordDuration ?? 0.6,
|
||||
font_size: options.fontSize ?? 60,
|
||||
music_volume: options.musicVolume ?? 0.25,
|
||||
@@ -91,6 +97,7 @@ export class FFmpegService {
|
||||
hasText: !!options.text,
|
||||
hasMusicId: !!options.musicId,
|
||||
hasMusicUrl: !!options.musicUrl,
|
||||
hasTTS: options.ttsEnabled,
|
||||
});
|
||||
|
||||
try {
|
||||
@@ -147,6 +154,8 @@ export class FFmpegService {
|
||||
text?: string;
|
||||
musicId?: string;
|
||||
musicUrl?: string;
|
||||
ttsEnabled?: boolean;
|
||||
ttsVoice?: string;
|
||||
wordDuration?: number;
|
||||
fontSize?: number;
|
||||
musicVolume?: number;
|
||||
@@ -159,6 +168,8 @@ export class FFmpegService {
|
||||
text: options.text,
|
||||
music_id: options.musicId,
|
||||
music_url: options.musicUrl,
|
||||
tts_enabled: options.ttsEnabled,
|
||||
tts_voice: options.ttsVoice,
|
||||
word_duration: options.wordDuration ?? 0.6,
|
||||
font_size: options.fontSize ?? 60,
|
||||
music_volume: options.musicVolume ?? 0.25,
|
||||
@@ -176,6 +187,7 @@ export class FFmpegService {
|
||||
hasText: !!options.text,
|
||||
hasMusicId: !!options.musicId,
|
||||
hasMusicUrl: !!options.musicUrl,
|
||||
hasTTS: options.ttsEnabled,
|
||||
});
|
||||
|
||||
try {
|
||||
|
||||
@@ -45,6 +45,13 @@ Ajout d'une fonctionnalité complète de création de Reels Facebook permettant
|
||||
|
||||
- [x] Configurer FreeSound API (env vars)
|
||||
- [x] Configurer FFmpeg API URL et clé (interface Settings)
|
||||
- [x] Ajouter la voix off (TTS) gratuite via Edge-TTS
|
||||
- [x] FFmpeg Service: Installer edge-tts et implémenter le mixage audio
|
||||
- [x] Backend: Supporter les options TTS (enable, voice)
|
||||
- [ ] Améliorer le TTS (Genre, Sync, Cleanup)
|
||||
- [ ] FFmpeg: Nettoyer le texte (No hashtags/emojis) avant TTS
|
||||
- [ ] FFmpeg: Générer fichier VTT pour synchro sous-titres
|
||||
- [ ] Frontend: Sélecteur de voix (Homme/Femme)
|
||||
- [/] Ajouter la valeur 'reel' à l'enum post_type en base de données
|
||||
- [x] Amélioration UI Musique (Pagination, Preview Audio)
|
||||
|
||||
|
||||
Reference in new issue
Block a user