From 37aba334252aad464b3fce92c65297742422a01b Mon Sep 17 00:00:00 2001 From: Michael SCHAL Date: Mon, 4 May 2026 13:13:55 +0200 Subject: [PATCH] fix: precise TTS word-boundary timing and remove wasted sync calculations - ffmpeg-service/main.py: replace ffsubsync path with exact word-boundary timing from edge_tts for TTS subtitles. Karaoke styling preserved. - server/services/ttsSync.ts: fix word count to match TTS-cleaned text, remove artificial punctuationPause subtraction, strip punctuation tokens from count. - server/routes/reels.ts: remove redundant ttsSyncService calls in preview/background; word_duration is ignored by Python, these only wasted TTS generations. Co-Authored-By: Claude Opus 4.7 --- ffmpeg-service/main.py | 47 ++++++++++++++++++++++++-------------- server/routes/reels.ts | 24 ++----------------- server/services/ttsSync.ts | 40 +++++++++----------------------- 3 files changed, 43 insertions(+), 68 deletions(-) diff --git a/ffmpeg-service/main.py b/ffmpeg-service/main.py index 86d1b2b..90ceef1 100644 --- a/ffmpeg-service/main.py +++ b/ffmpeg-service/main.py @@ -291,16 +291,16 @@ async def generate_tts_with_subs( # Use display_text for subtitle content if provided text_to_display = display_text if display_text else text - print("⚠️ Using ffsubsync for precise synchronization as requested by user.") - unsynced_srt_path = audio_path.with_suffix(".unsynced.srt") - synced_srt_path = audio_path.with_suffix(".synced.srt") - - # Subtitle syncing pipeline using ffsubsync for the generated TTS voice - generate_unsynced_srt(text_to_display, unsynced_srt_path, total_duration=audio_duration) - run_ffsubsync(audio_path, unsynced_srt_path, synced_srt_path) - convert_srt_to_ass(synced_srt_path, ass_path, font_size=65, delay=delay) - - print(f"✅ TTS synchronisation completed with ffsubsync") + print("🎯 Using precise word-boundary timing from TTS engine") + generate_ass_from_word_boundaries( + word_boundaries, + text_to_display, + ass_path, + font_size=65, + total_duration=audio_duration, + delay=delay, + ) + print(f"✅ TTS synchronisation completed with word-boundary timing") return else: print(f"⚠️ Audio file empty or missing with voice: {attempt_voice}") @@ -322,19 +322,22 @@ def generate_ass_from_word_boundaries( ): """Generate ASS subtitles using precise word-level timing from TTS engine. - Groups words into readable chunks (~5 words or at punctuation) and uses + Groups words into readable chunks (~3 words or at punctuation) and uses the real start/end timestamps from the TTS engine for each chunk. + Includes karaoke fill tags for word-level visual highlighting. """ if not word_boundaries: return - # Group word boundaries into chunks of ~5 words, or split at punctuation + # Group word boundaries into chunks of ~3 words, or split at punctuation chunks = [] current_words = [] + current_boundaries = [] current_start = word_boundaries[0]["offset"] for i, wb in enumerate(word_boundaries): current_words.append(wb["text"]) + current_boundaries.append(wb) is_last = i == len(word_boundaries) - 1 # Split at punctuation or every 3 words (tighter sync with voice) ends_sentence = wb["text"].rstrip().endswith((".", "!", "?", ":", ",")) @@ -345,12 +348,14 @@ def generate_ass_from_word_boundaries( chunk_end = wb["offset"] + wb["duration"] chunks.append( { - "text": " ".join(current_words), + "words": current_words, + "boundaries": current_boundaries, "start": current_start + delay, "end": chunk_end + delay, } ) current_words = [] + current_boundaries = [] # Next chunk starts at the next word's offset if not is_last: current_start = word_boundaries[i + 1]["offset"] @@ -364,7 +369,7 @@ def generate_ass_from_word_boundaries( for i in range(len(chunks) - 1): chunks[i]["end"] = max(chunks[i]["end"], chunks[i + 1]["start"] + 0.05) - # ASS Header + # ASS Header (same karaoke style as convert_srt_to_ass) header = f"""[Script Info] ScriptType: v4.00+ PlayResX: 1080 @@ -373,7 +378,7 @@ ScaledBorderAndShadow: yes [V4+ Styles] Format: Name, Fontname, Fontsize, PrimaryColour, SecondaryColour, OutlineColour, BackColour, Bold, Italic, Underline, StrikeOut, ScaleX, ScaleY, Spacing, Angle, BorderStyle, Outline, Shadow, Alignment, MarginL, MarginR, MarginV, Encoding -Style: Default,Sans,{font_size},&H00FFFFFF,&H000000FF,&H00000000,&H80000000,-1,0,0,0,100,100,0,0,1,4,2,5,50,50,0,1 +Style: Default,Sans,{font_size},&H0000FFFF,&H00FFFFFF,&H00000000,&H80000000,-1,0,0,0,100,100,0,0,1,4,2,5,50,50,0,1 [Events] Format: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text @@ -383,8 +388,16 @@ Format: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text for chunk in chunks: start_ts = format_ass_time(chunk["start"]) end_ts = format_ass_time(chunk["end"]) - sanitized = chunk["text"].replace("{", "(").replace("}", ")") - events += f"Dialogue: 0,{start_ts},{end_ts},Default,,0,0,0,,{sanitized}\n" + + karaoke_parts = [] + for wb in chunk["boundaries"]: + # duration in centiseconds, minimum 10cs to avoid zero + w_dur_cs = max(10, int(wb["duration"] * 100)) + sanitized = wb["text"].replace("{", "(").replace("}", ")") + karaoke_parts.append(f"{{\\kf{w_dur_cs}}}{sanitized}") + + karaoke_text = " ".join(karaoke_parts) + events += f"Dialogue: 0,{start_ts},{end_ts},Default,,0,0,0,,{karaoke_text}\n" with open(ass_path, "w", encoding="utf-8") as f: f.write(header + events) diff --git a/server/routes/reels.ts b/server/routes/reels.ts index df716ed..61bf3ea 100644 --- a/server/routes/reels.ts +++ b/server/routes/reels.ts @@ -350,17 +350,7 @@ reelsRouter.post('/reels/preview', async (req: Request, res: Response) => { console.error('Error fetching watermark configuration', e); } - // Calculer le word_duration synchronisé si TTS activé - let finalWordDuration = wordDuration; - if (ttsEnabled && overlayText) { - try { - const sync = await ttsSyncService.calculateSyncTiming(overlayText, ttsVoice); - finalWordDuration = sync.wordDuration; - console.log(`🎯 TTS Sync (preview): ${sync.wordCount} mots, ${sync.audioDuration.toFixed(2)}s audio, word_duration=${finalWordDuration.toFixed(3)}s`); - } catch (syncErr) { - console.warn('⚠️ TTS sync failed (preview), using default word_duration:', syncErr); - } - } + const finalWordDuration = wordDuration; // Traiter la vidéo via FFmpeg const result = await ffmpegService.processReelFromUrl(resolveInternalUrl(media.originalUrl), { @@ -593,17 +583,7 @@ async function processReelBackground( console.error('Error fetching watermark configuration', e); } - // Calculer le word_duration synchronisé si TTS activé - let finalWordDuration = wordDuration ?? 0.6; - if (ttsEnabled && overlayText && ttsVoice) { - try { - const sync = await ttsSyncService.calculateSyncTiming(overlayText, ttsVoice); - finalWordDuration = sync.wordDuration; - console.log(`🎯 TTS Sync (background): ${sync.wordCount} mots, ${sync.audioDuration.toFixed(2)}s audio, word_duration=${finalWordDuration.toFixed(3)}s`); - } catch (syncErr) { - console.warn('⚠️ TTS sync failed (background), using default word_duration:', syncErr); - } - } + const finalWordDuration = wordDuration ?? 0.6; const ffmpegResult = await ffmpegService.processReelFromUrl(resolveInternalUrl(media.originalUrl), { text: overlayText, diff --git a/server/services/ttsSync.ts b/server/services/ttsSync.ts index 0dc6395..e8fe3d0 100644 --- a/server/services/ttsSync.ts +++ b/server/services/ttsSync.ts @@ -16,11 +16,10 @@ export class TtsSyncService { * avec la durée réelle de la voix TTS générée. */ async calculateSyncTiming(text: string, voice: string): Promise { - const cleanTtsText = this.cleanForTts(text); - const cleanDisplayText = this.cleanForDisplay(text); + const cleanText = this.cleanText(text); // 1. Générer le TTS preview et mesurer sa durée exacte - const ttsResult = await ffmpegService.previewTTS(cleanTtsText, voice); + const ttsResult = await ffmpegService.previewTTS(cleanText, voice); if (!ttsResult.success || !ttsResult.audioBase64) { throw new Error('TTS preview failed: ' + (ttsResult.error || 'unknown')); } @@ -29,14 +28,11 @@ export class TtsSyncService { const metadata = await musicMetadata.parseBuffer(audioBuffer, 'audio/mpeg'); const audioDuration = metadata.format.duration || 0; - // 2. Analyser le texte - const words = cleanDisplayText.split(/\s+/).filter(w => w.length > 0); - const wordCount = words.length; - const punctuationPause = this.calculatePunctuationPauses(cleanDisplayText); + // 2. Analyser le texte (compte les mots réellement lus par la voix) + const wordCount = this.calculateWordCount(cleanText); - // 3. Calculer le word_duration ajusté - const effectiveDuration = Math.max(audioDuration - punctuationPause, 0.5); - const wordDuration = wordCount > 0 ? effectiveDuration / wordCount : 0.6; + // 3. Calculer le word_duration + const wordDuration = wordCount > 0 ? audioDuration / wordCount : 0.6; // 4. Validation const warnings: string[] = []; @@ -52,13 +48,13 @@ export class TtsSyncService { wordDuration, audioDuration, wordCount, - punctuationPause, + punctuationPause: 0, isHealthy, warnings, }; } - private cleanForTts(text: string): string { + private cleanText(text: string): string { return text .replace(/#\w+/g, '') .replace(/https?:\/\/\S+/g, '') @@ -66,23 +62,9 @@ export class TtsSyncService { .trim(); } - private cleanForDisplay(text: string): string { - return text.trim(); - } - - private calculatePunctuationPauses(text: string): number { - let pause = 0; - const chars = text.split(''); - for (const c of chars) { - if (',;'.includes(c)) pause += 0.3; - else if ('.!?'.includes(c)) pause += 0.6; - else if (':'.includes(c)) pause += 0.4; - else if (c === '\n') pause += 0.3; - } - // "..." compte comme un seul point mais pause plus longue - const ellipsisCount = (text.match(/\.\.\./g) || []).length; - pause += ellipsisCount * 0.3; // bonus pour les points de suspension - return pause; + private calculateWordCount(text: string): number { + const tokens = text.split(/\s+/).filter(w => w.length > 0); + return tokens.filter(w => /[a-zA-Z0-9À-ſ]/.test(w)).length; } }