From 32c308c2af526b15a39e32bab8e5c982ec9dbd66 Mon Sep 17 00:00:00 2001 From: Michael SCHAL Date: Tue, 24 Mar 2026 19:22:28 +0100 Subject: [PATCH] fix: use exact MP3 duration for word-timing sync instead of byte estimate Parses the generated TTS audio file with music-metadata to get the real duration before computing word timings, eliminating desync between text overlay and voice playback. Co-Authored-By: Claude Sonnet 4.6 --- server/routes/remotion.ts | 16 ++++++++++++---- 1 file changed, 12 insertions(+), 4 deletions(-) diff --git a/server/routes/remotion.ts b/server/routes/remotion.ts index a5e6325..b4481be 100644 --- a/server/routes/remotion.ts +++ b/server/routes/remotion.ts @@ -6,6 +6,7 @@ import { bundle } from "@remotion/bundler"; import { renderMedia, selectComposition } from "@remotion/renderer"; import { ffmpegService } from "../services/ffmpeg"; import { storage as dbStorage } from "../storage"; +import * as musicMetadata from "music-metadata"; export const remotionRouter = Router(); @@ -158,12 +159,19 @@ remotionRouter.post("/render", upload.fields([{ name: "images", maxCount: 4 }, { if (ttsResult.success && ttsResult.audioBase64) { const audioFilename = `tts-${Date.now()}.mp3`; const audioPath = path.join(uploadDir, audioFilename); - fs.writeFileSync(audioPath, Buffer.from(ttsResult.audioBase64, "base64")); + const audioBuffer = Buffer.from(ttsResult.audioBase64, "base64"); + fs.writeFileSync(audioPath, audioBuffer); audioUrl = `${host}/uploads/temp/${audioFilename}`; - // Estimate audio duration (MP3 at 128kbps ~ 16KB/s) - const audioBytes = Buffer.from(ttsResult.audioBase64, "base64").length; - estimatedAudioDuration = Math.max(audioBytes / 16000, ttsText.split(/\s+/).length * 0.5); + // Get exact audio duration from MP3 metadata + try { + const meta = await musicMetadata.parseFile(audioPath); + estimatedAudioDuration = meta.format.duration ?? 0; + } catch { + // Fallback: rough estimate from byte size (128kbps = 16KB/s) + estimatedAudioDuration = audioBuffer.length / 16000; + } + estimatedAudioDuration = Math.max(estimatedAudioDuration, ttsText.split(/\s+/).length * 0.35); wordTimings = computeWordTimings(overlayText, estimatedAudioDuration, FPS, 0); console.log(`✅ TTS generated, ~${estimatedAudioDuration.toFixed(1)}s, ${wordTimings.length} words`);