mirror of
https://github.com/R0m1k3/Socialflow.git
synced 2026-10-11 17:26:45 +02:00
fix: precise TTS word-boundary timing and remove wasted sync calculations
- ffmpeg-service/main.py: replace ffsubsync path with exact word-boundary timing from edge_tts for TTS subtitles. Karaoke styling preserved. - server/services/ttsSync.ts: fix word count to match TTS-cleaned text, remove artificial punctuationPause subtraction, strip punctuation tokens from count. - server/routes/reels.ts: remove redundant ttsSyncService calls in preview/background; word_duration is ignored by Python, these only wasted TTS generations. Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
1 parent
9ce735dc23
commit
37aba33425
3 files changed
+43
-68
No files matched your search
+30
-17
@@ -291,16 +291,16 @@ async def generate_tts_with_subs(
|
||||
# Use display_text for subtitle content if provided
|
||||
text_to_display = display_text if display_text else text
|
||||
|
||||
print("⚠️ Using ffsubsync for precise synchronization as requested by user.")
|
||||
unsynced_srt_path = audio_path.with_suffix(".unsynced.srt")
|
||||
synced_srt_path = audio_path.with_suffix(".synced.srt")
|
||||
|
||||
# Subtitle syncing pipeline using ffsubsync for the generated TTS voice
|
||||
generate_unsynced_srt(text_to_display, unsynced_srt_path, total_duration=audio_duration)
|
||||
run_ffsubsync(audio_path, unsynced_srt_path, synced_srt_path)
|
||||
convert_srt_to_ass(synced_srt_path, ass_path, font_size=65, delay=delay)
|
||||
|
||||
print(f"✅ TTS synchronisation completed with ffsubsync")
|
||||
print("🎯 Using precise word-boundary timing from TTS engine")
|
||||
generate_ass_from_word_boundaries(
|
||||
word_boundaries,
|
||||
text_to_display,
|
||||
ass_path,
|
||||
font_size=65,
|
||||
total_duration=audio_duration,
|
||||
delay=delay,
|
||||
)
|
||||
print(f"✅ TTS synchronisation completed with word-boundary timing")
|
||||
return
|
||||
else:
|
||||
print(f"⚠️ Audio file empty or missing with voice: {attempt_voice}")
|
||||
@@ -322,19 +322,22 @@ def generate_ass_from_word_boundaries(
|
||||
):
|
||||
"""Generate ASS subtitles using precise word-level timing from TTS engine.
|
||||
|
||||
Groups words into readable chunks (~5 words or at punctuation) and uses
|
||||
Groups words into readable chunks (~3 words or at punctuation) and uses
|
||||
the real start/end timestamps from the TTS engine for each chunk.
|
||||
Includes karaoke fill tags for word-level visual highlighting.
|
||||
"""
|
||||
if not word_boundaries:
|
||||
return
|
||||
|
||||
# Group word boundaries into chunks of ~5 words, or split at punctuation
|
||||
# Group word boundaries into chunks of ~3 words, or split at punctuation
|
||||
chunks = []
|
||||
current_words = []
|
||||
current_boundaries = []
|
||||
current_start = word_boundaries[0]["offset"]
|
||||
|
||||
for i, wb in enumerate(word_boundaries):
|
||||
current_words.append(wb["text"])
|
||||
current_boundaries.append(wb)
|
||||
is_last = i == len(word_boundaries) - 1
|
||||
# Split at punctuation or every 3 words (tighter sync with voice)
|
||||
ends_sentence = wb["text"].rstrip().endswith((".", "!", "?", ":", ","))
|
||||
@@ -345,12 +348,14 @@ def generate_ass_from_word_boundaries(
|
||||
chunk_end = wb["offset"] + wb["duration"]
|
||||
chunks.append(
|
||||
{
|
||||
"text": " ".join(current_words),
|
||||
"words": current_words,
|
||||
"boundaries": current_boundaries,
|
||||
"start": current_start + delay,
|
||||
"end": chunk_end + delay,
|
||||
}
|
||||
)
|
||||
current_words = []
|
||||
current_boundaries = []
|
||||
# Next chunk starts at the next word's offset
|
||||
if not is_last:
|
||||
current_start = word_boundaries[i + 1]["offset"]
|
||||
@@ -364,7 +369,7 @@ def generate_ass_from_word_boundaries(
|
||||
for i in range(len(chunks) - 1):
|
||||
chunks[i]["end"] = max(chunks[i]["end"], chunks[i + 1]["start"] + 0.05)
|
||||
|
||||
# ASS Header
|
||||
# ASS Header (same karaoke style as convert_srt_to_ass)
|
||||
header = f"""[Script Info]
|
||||
ScriptType: v4.00+
|
||||
PlayResX: 1080
|
||||
@@ -373,7 +378,7 @@ ScaledBorderAndShadow: yes
|
||||
|
||||
[V4+ Styles]
|
||||
Format: Name, Fontname, Fontsize, PrimaryColour, SecondaryColour, OutlineColour, BackColour, Bold, Italic, Underline, StrikeOut, ScaleX, ScaleY, Spacing, Angle, BorderStyle, Outline, Shadow, Alignment, MarginL, MarginR, MarginV, Encoding
|
||||
Style: Default,Sans,{font_size},&H00FFFFFF,&H000000FF,&H00000000,&H80000000,-1,0,0,0,100,100,0,0,1,4,2,5,50,50,0,1
|
||||
Style: Default,Sans,{font_size},&H0000FFFF,&H00FFFFFF,&H00000000,&H80000000,-1,0,0,0,100,100,0,0,1,4,2,5,50,50,0,1
|
||||
|
||||
[Events]
|
||||
Format: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text
|
||||
@@ -383,8 +388,16 @@ Format: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text
|
||||
for chunk in chunks:
|
||||
start_ts = format_ass_time(chunk["start"])
|
||||
end_ts = format_ass_time(chunk["end"])
|
||||
sanitized = chunk["text"].replace("{", "(").replace("}", ")")
|
||||
events += f"Dialogue: 0,{start_ts},{end_ts},Default,,0,0,0,,{sanitized}\n"
|
||||
|
||||
karaoke_parts = []
|
||||
for wb in chunk["boundaries"]:
|
||||
# duration in centiseconds, minimum 10cs to avoid zero
|
||||
w_dur_cs = max(10, int(wb["duration"] * 100))
|
||||
sanitized = wb["text"].replace("{", "(").replace("}", ")")
|
||||
karaoke_parts.append(f"{{\\kf{w_dur_cs}}}{sanitized}")
|
||||
|
||||
karaoke_text = " ".join(karaoke_parts)
|
||||
events += f"Dialogue: 0,{start_ts},{end_ts},Default,,0,0,0,,{karaoke_text}\n"
|
||||
|
||||
with open(ass_path, "w", encoding="utf-8") as f:
|
||||
f.write(header + events)
|
||||
|
||||
+2
-22
@@ -350,17 +350,7 @@ reelsRouter.post('/reels/preview', async (req: Request, res: Response) => {
|
||||
console.error('Error fetching watermark configuration', e);
|
||||
}
|
||||
|
||||
// Calculer le word_duration synchronisé si TTS activé
|
||||
let finalWordDuration = wordDuration;
|
||||
if (ttsEnabled && overlayText) {
|
||||
try {
|
||||
const sync = await ttsSyncService.calculateSyncTiming(overlayText, ttsVoice);
|
||||
finalWordDuration = sync.wordDuration;
|
||||
console.log(`🎯 TTS Sync (preview): ${sync.wordCount} mots, ${sync.audioDuration.toFixed(2)}s audio, word_duration=${finalWordDuration.toFixed(3)}s`);
|
||||
} catch (syncErr) {
|
||||
console.warn('⚠️ TTS sync failed (preview), using default word_duration:', syncErr);
|
||||
}
|
||||
}
|
||||
const finalWordDuration = wordDuration;
|
||||
|
||||
// Traiter la vidéo via FFmpeg
|
||||
const result = await ffmpegService.processReelFromUrl(resolveInternalUrl(media.originalUrl), {
|
||||
@@ -593,17 +583,7 @@ async function processReelBackground(
|
||||
console.error('Error fetching watermark configuration', e);
|
||||
}
|
||||
|
||||
// Calculer le word_duration synchronisé si TTS activé
|
||||
let finalWordDuration = wordDuration ?? 0.6;
|
||||
if (ttsEnabled && overlayText && ttsVoice) {
|
||||
try {
|
||||
const sync = await ttsSyncService.calculateSyncTiming(overlayText, ttsVoice);
|
||||
finalWordDuration = sync.wordDuration;
|
||||
console.log(`🎯 TTS Sync (background): ${sync.wordCount} mots, ${sync.audioDuration.toFixed(2)}s audio, word_duration=${finalWordDuration.toFixed(3)}s`);
|
||||
} catch (syncErr) {
|
||||
console.warn('⚠️ TTS sync failed (background), using default word_duration:', syncErr);
|
||||
}
|
||||
}
|
||||
const finalWordDuration = wordDuration ?? 0.6;
|
||||
|
||||
const ffmpegResult = await ffmpegService.processReelFromUrl(resolveInternalUrl(media.originalUrl), {
|
||||
text: overlayText,
|
||||
|
||||
+11
-29
@@ -16,11 +16,10 @@ export class TtsSyncService {
|
||||
* avec la durée réelle de la voix TTS générée.
|
||||
*/
|
||||
async calculateSyncTiming(text: string, voice: string): Promise<SyncTiming> {
|
||||
const cleanTtsText = this.cleanForTts(text);
|
||||
const cleanDisplayText = this.cleanForDisplay(text);
|
||||
const cleanText = this.cleanText(text);
|
||||
|
||||
// 1. Générer le TTS preview et mesurer sa durée exacte
|
||||
const ttsResult = await ffmpegService.previewTTS(cleanTtsText, voice);
|
||||
const ttsResult = await ffmpegService.previewTTS(cleanText, voice);
|
||||
if (!ttsResult.success || !ttsResult.audioBase64) {
|
||||
throw new Error('TTS preview failed: ' + (ttsResult.error || 'unknown'));
|
||||
}
|
||||
@@ -29,14 +28,11 @@ export class TtsSyncService {
|
||||
const metadata = await musicMetadata.parseBuffer(audioBuffer, 'audio/mpeg');
|
||||
const audioDuration = metadata.format.duration || 0;
|
||||
|
||||
// 2. Analyser le texte
|
||||
const words = cleanDisplayText.split(/\s+/).filter(w => w.length > 0);
|
||||
const wordCount = words.length;
|
||||
const punctuationPause = this.calculatePunctuationPauses(cleanDisplayText);
|
||||
// 2. Analyser le texte (compte les mots réellement lus par la voix)
|
||||
const wordCount = this.calculateWordCount(cleanText);
|
||||
|
||||
// 3. Calculer le word_duration ajusté
|
||||
const effectiveDuration = Math.max(audioDuration - punctuationPause, 0.5);
|
||||
const wordDuration = wordCount > 0 ? effectiveDuration / wordCount : 0.6;
|
||||
// 3. Calculer le word_duration
|
||||
const wordDuration = wordCount > 0 ? audioDuration / wordCount : 0.6;
|
||||
|
||||
// 4. Validation
|
||||
const warnings: string[] = [];
|
||||
@@ -52,13 +48,13 @@ export class TtsSyncService {
|
||||
wordDuration,
|
||||
audioDuration,
|
||||
wordCount,
|
||||
punctuationPause,
|
||||
punctuationPause: 0,
|
||||
isHealthy,
|
||||
warnings,
|
||||
};
|
||||
}
|
||||
|
||||
private cleanForTts(text: string): string {
|
||||
private cleanText(text: string): string {
|
||||
return text
|
||||
.replace(/#\w+/g, '')
|
||||
.replace(/https?:\/\/\S+/g, '')
|
||||
@@ -66,23 +62,9 @@ export class TtsSyncService {
|
||||
.trim();
|
||||
}
|
||||
|
||||
private cleanForDisplay(text: string): string {
|
||||
return text.trim();
|
||||
}
|
||||
|
||||
private calculatePunctuationPauses(text: string): number {
|
||||
let pause = 0;
|
||||
const chars = text.split('');
|
||||
for (const c of chars) {
|
||||
if (',;'.includes(c)) pause += 0.3;
|
||||
else if ('.!?'.includes(c)) pause += 0.6;
|
||||
else if (':'.includes(c)) pause += 0.4;
|
||||
else if (c === '\n') pause += 0.3;
|
||||
}
|
||||
// "..." compte comme un seul point mais pause plus longue
|
||||
const ellipsisCount = (text.match(/\.\.\./g) || []).length;
|
||||
pause += ellipsisCount * 0.3; // bonus pour les points de suspension
|
||||
return pause;
|
||||
private calculateWordCount(text: string): number {
|
||||
const tokens = text.split(/\s+/).filter(w => w.length > 0);
|
||||
return tokens.filter(w => /[a-zA-Z0-9À-ſ]/.test(w)).length;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Reference in new issue
Block a user