mirror of
https://github.com/R0m1k3/Socialflow.git
synced 2026-10-11 17:26:45 +02:00
Fins de phrases coupées : - Le fondu final ne commence plus avant la fin de la voix (il rognait la dernière phrase quand la vidéo finissait peu après) ; il se raccourcit au besoin. Même règle côté Python, aperçu et rendu Remotion - SRT : découpe au milieu de la zone la plus calme de chaque pause, mesurée sur l'enveloppe sonore (tranches de 10 ms), avec un seuil relatif au niveau de la voix qui respecte les fins douces ; 300 ms gardées après le dernier son Bruits parasites : - Normalisation de la voix en deux passes, en mode linéaire : le mode dynamique relevait souffles et bruits entre les phrases (mesuré : -64 dB avant, -75 dB après dans les pauses) - Réduction de bruit douce (afftdn), compresseur sans gain ajouté, fondus de 5 ms aux extrémités ; mix final moins « pompant » (LRA 20) - Voix Gemini vérifiée : les bruits ajoutés avant le premier ou après le dernier mot sont retirés Vérification de lecture comparée lettre à lettre : les petites erreurs de Whisper ne déclenchent plus de régénérations inutiles. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_018Ze4bs7tpF1KGWUk6ZZSZ4
325 lines
12 KiB
TypeScript
325 lines
12 KiB
TypeScript
/**
|
|
* Client du service FFmpeg (conteneur Python `ffmpeg-service`).
|
|
*
|
|
* Le service rend le Reel puis expose le MP4 en téléchargement
|
|
* (GET /files/{job}/output.mp4) : la vidéo ne transite plus en base64 dans du
|
|
* JSON, qui gonflait sa taille d'un tiers et la gardait entière en mémoire.
|
|
*/
|
|
|
|
import fs from 'fs';
|
|
import { Readable } from 'stream';
|
|
import { pipeline } from 'stream/promises';
|
|
import type { ReadableStream as WebReadableStream } from 'stream/web';
|
|
import type { SrtCue } from '@shared/srt';
|
|
import type { TtsEngine, TtsStyle } from '@shared/voices';
|
|
|
|
/** Un rendu long (stabilisation + encodage) peut dépasser plusieurs minutes. */
|
|
const PROCESS_TIMEOUT_MS = 20 * 60_000;
|
|
const DOWNLOAD_TIMEOUT_MS = 5 * 60_000;
|
|
const TTS_TIMEOUT_MS = 3 * 60_000;
|
|
const HEALTH_TIMEOUT_MS = 5_000;
|
|
|
|
export interface ReelRenderOptions {
|
|
text?: string;
|
|
/** Sous-titres SRT : remplacent `text`, la voix lit chacun à son instant. */
|
|
srtCues?: SrtCue[];
|
|
musicUrl?: string;
|
|
ttsEnabled?: boolean;
|
|
ttsVoice?: string;
|
|
ttsEngine?: TtsEngine;
|
|
ttsStyle?: TtsStyle;
|
|
geminiApiKey?: string;
|
|
fontSize?: number;
|
|
musicVolume?: number;
|
|
drawText?: boolean;
|
|
stabilize?: boolean;
|
|
watermarkUrl?: string;
|
|
storeName?: string;
|
|
enableEndingEffect?: boolean;
|
|
/** Petit logo pendant la vidéo (le grand logo de fin dépend de enableEndingEffect). */
|
|
showLogo?: boolean;
|
|
}
|
|
|
|
export interface ReelRenderResult {
|
|
video: Buffer;
|
|
duration: number;
|
|
ttsEngine?: string | null;
|
|
ttsVoice?: string | null;
|
|
warnings: string[];
|
|
}
|
|
|
|
/** Rendu préparé pour Remotion : fichiers déjà écrits sur le disque. */
|
|
export interface PreparedReel {
|
|
videoPath: string;
|
|
audioPath: string | null;
|
|
totalDuration: number;
|
|
videoDuration: number;
|
|
logoStart: number | null;
|
|
/** Fondu final, calculé par le service pour ne jamais rogner la voix. */
|
|
fadeStart: number | null;
|
|
fadeDuration: number | null;
|
|
words: TimedWord[];
|
|
ttsEngine?: string | null;
|
|
ttsVoice?: string | null;
|
|
warnings: string[];
|
|
}
|
|
|
|
export interface TimedWord {
|
|
text: string;
|
|
start: number;
|
|
end: number;
|
|
}
|
|
|
|
export interface VoicePreview {
|
|
audio: Buffer;
|
|
duration: number;
|
|
words: TimedWord[];
|
|
engine: string;
|
|
voice: string;
|
|
warnings: string[];
|
|
}
|
|
|
|
interface FFmpegConfig {
|
|
apiUrl: string;
|
|
apiKey: string;
|
|
}
|
|
|
|
/** Erreur renvoyée par le service, avec son message lisible. */
|
|
export class FFmpegServiceError extends Error {
|
|
constructor(message: string, readonly status?: number) {
|
|
super(message);
|
|
this.name = 'FFmpegServiceError';
|
|
}
|
|
}
|
|
|
|
export class FFmpegService {
|
|
private config: FFmpegConfig | null = null;
|
|
|
|
configure(apiUrl: string, apiKey: string): void {
|
|
this.config = { apiUrl: apiUrl.replace(/\/$/, ''), apiKey };
|
|
console.log('🎬 FFmpeg Service configured:', this.config.apiUrl);
|
|
}
|
|
|
|
private ensureConfigured(): FFmpegConfig {
|
|
if (!this.config) {
|
|
throw new FFmpegServiceError("Le service FFmpeg n'est pas configuré (FFMPEG_API_URL / FFMPEG_API_KEY).");
|
|
}
|
|
return this.config;
|
|
}
|
|
|
|
private async call(path: string, init: RequestInit & { timeoutMs: number }): Promise<Response> {
|
|
const config = this.ensureConfigured();
|
|
const { timeoutMs, headers, ...rest } = init;
|
|
const response = await fetch(`${config.apiUrl}${path}`, {
|
|
...rest,
|
|
headers: { 'X-API-Key': config.apiKey, ...headers },
|
|
signal: AbortSignal.timeout(timeoutMs),
|
|
});
|
|
if (!response.ok) {
|
|
const body = await response.text();
|
|
let detail = body;
|
|
try {
|
|
detail = JSON.parse(body).detail ?? body;
|
|
} catch { /* corps non JSON */ }
|
|
throw new FFmpegServiceError(String(detail).slice(0, 2000), response.status);
|
|
}
|
|
return response;
|
|
}
|
|
|
|
/**
|
|
* Rend un Reel à partir de l'URL d'une vidéo et renvoie le MP4 produit.
|
|
* Lève FFmpegServiceError en cas d'échec (voix comprise).
|
|
*/
|
|
async renderReel(videoUrl: string, options: ReelRenderOptions = {}): Promise<ReelRenderResult> {
|
|
const body = {
|
|
video_url: videoUrl,
|
|
text: options.text,
|
|
srt_cues: options.srtCues,
|
|
music_url: options.musicUrl,
|
|
tts_enabled: options.ttsEnabled ?? false,
|
|
tts_voice: options.ttsVoice,
|
|
tts_engine: options.ttsEngine,
|
|
tts_style: options.ttsStyle,
|
|
gemini_api_key: options.geminiApiKey,
|
|
font_size: options.fontSize ?? 64,
|
|
music_volume: options.musicVolume ?? 0.25,
|
|
draw_text: options.drawText ?? true,
|
|
stabilize: options.stabilize ?? false,
|
|
watermark_url: options.watermarkUrl,
|
|
store_name: options.storeName,
|
|
enable_ending_effect: options.enableEndingEffect ?? true,
|
|
show_watermark: options.showLogo ?? true,
|
|
};
|
|
|
|
console.log('🎬 Rendu du Reel :', {
|
|
videoUrl,
|
|
textLength: options.text?.length ?? 0,
|
|
music: !!options.musicUrl,
|
|
tts: options.ttsEnabled ? `${options.ttsEngine}/${options.ttsVoice}/${options.ttsStyle ?? 'neutral'}` : false,
|
|
});
|
|
|
|
const response = await this.call('/process-reel', {
|
|
method: 'POST',
|
|
headers: { 'Content-Type': 'application/json' },
|
|
body: JSON.stringify(body),
|
|
timeoutMs: PROCESS_TIMEOUT_MS,
|
|
});
|
|
const data = await response.json() as {
|
|
job_id: string;
|
|
output_path: string;
|
|
duration: number;
|
|
tts_engine?: string | null;
|
|
tts_voice?: string | null;
|
|
warnings?: string[];
|
|
};
|
|
|
|
try {
|
|
const file = await this.call(data.output_path, { method: 'GET', timeoutMs: DOWNLOAD_TIMEOUT_MS });
|
|
const video = Buffer.from(await file.arrayBuffer());
|
|
for (const warning of data.warnings ?? []) console.warn(`⚠️ [FFmpeg] ${warning}`);
|
|
return {
|
|
video,
|
|
duration: data.duration,
|
|
ttsEngine: data.tts_engine,
|
|
ttsVoice: data.tts_voice,
|
|
warnings: data.warnings ?? [],
|
|
};
|
|
} finally {
|
|
// Le fichier n'est plus utile au service une fois récupéré
|
|
this.call(`/jobs/${data.job_id}`, { method: 'DELETE', timeoutMs: HEALTH_TIMEOUT_MS })
|
|
.catch(() => { /* purgé plus tard par le service */ });
|
|
}
|
|
}
|
|
|
|
/**
|
|
* Prépare un rendu Remotion : image recadrée à la durée finale et piste son
|
|
* finale, écrites dans `targetDir`, plus le minutage des mots.
|
|
*/
|
|
async prepareReel(
|
|
videoUrl: string,
|
|
options: ReelRenderOptions & { hasLogo: boolean },
|
|
targetDir: string,
|
|
): Promise<PreparedReel> {
|
|
const response = await this.call('/prepare-reel', {
|
|
method: 'POST',
|
|
headers: { 'Content-Type': 'application/json' },
|
|
body: JSON.stringify({
|
|
video_url: videoUrl,
|
|
text: options.text,
|
|
srt_cues: options.srtCues,
|
|
music_url: options.musicUrl,
|
|
tts_enabled: options.ttsEnabled ?? false,
|
|
tts_voice: options.ttsVoice,
|
|
tts_engine: options.ttsEngine,
|
|
tts_style: options.ttsStyle,
|
|
gemini_api_key: options.geminiApiKey,
|
|
music_volume: options.musicVolume ?? 0.25,
|
|
draw_text: options.drawText ?? true,
|
|
stabilize: options.stabilize ?? false,
|
|
store_name: options.storeName,
|
|
enable_ending_effect: options.enableEndingEffect ?? true,
|
|
has_logo: options.hasLogo,
|
|
}),
|
|
timeoutMs: PROCESS_TIMEOUT_MS,
|
|
});
|
|
const data = await response.json() as {
|
|
job_id: string;
|
|
video_path: string;
|
|
audio_path: string | null;
|
|
total_duration: number;
|
|
video_duration: number;
|
|
logo_start: number | null;
|
|
fade_start?: number | null;
|
|
fade_duration?: number | null;
|
|
words: TimedWord[];
|
|
tts_engine?: string | null;
|
|
tts_voice?: string | null;
|
|
warnings?: string[];
|
|
};
|
|
|
|
try {
|
|
await fs.promises.mkdir(targetDir, { recursive: true });
|
|
const videoPath = `${targetDir}/video.mp4`;
|
|
await this.downloadTo(data.video_path, videoPath);
|
|
let audioPath: string | null = null;
|
|
if (data.audio_path) {
|
|
audioPath = `${targetDir}/audio.wav`;
|
|
await this.downloadTo(data.audio_path, audioPath);
|
|
}
|
|
for (const warning of data.warnings ?? []) console.warn(`⚠️ [FFmpeg] ${warning}`);
|
|
return {
|
|
videoPath,
|
|
audioPath,
|
|
totalDuration: data.total_duration,
|
|
videoDuration: data.video_duration,
|
|
logoStart: data.logo_start,
|
|
fadeStart: data.fade_start ?? null,
|
|
fadeDuration: data.fade_duration ?? null,
|
|
words: data.words ?? [],
|
|
ttsEngine: data.tts_engine,
|
|
ttsVoice: data.tts_voice,
|
|
warnings: data.warnings ?? [],
|
|
};
|
|
} finally {
|
|
this.call(`/jobs/${data.job_id}`, { method: 'DELETE', timeoutMs: HEALTH_TIMEOUT_MS })
|
|
.catch(() => { /* purgé plus tard par le service */ });
|
|
}
|
|
}
|
|
|
|
/** Télécharge un fichier du service en flux, sans le charger en mémoire. */
|
|
private async downloadTo(servicePath: string, target: string): Promise<void> {
|
|
const response = await this.call(servicePath, { method: 'GET', timeoutMs: DOWNLOAD_TIMEOUT_MS });
|
|
if (!response.body) throw new FFmpegServiceError(`Réponse vide pour ${servicePath}`);
|
|
await pipeline(
|
|
Readable.fromWeb(response.body as unknown as WebReadableStream<Uint8Array>),
|
|
fs.createWriteStream(target),
|
|
);
|
|
}
|
|
|
|
async healthCheck(): Promise<boolean> {
|
|
try {
|
|
await this.call('/health', { method: 'GET', timeoutMs: HEALTH_TIMEOUT_MS });
|
|
return true;
|
|
} catch {
|
|
return false;
|
|
}
|
|
}
|
|
|
|
/** Génère la voix seule (aperçu), avec le minutage de chaque mot. */
|
|
async previewVoice(
|
|
text: string,
|
|
options: { voice?: string; engine?: TtsEngine; style?: TtsStyle; geminiApiKey?: string } = {},
|
|
): Promise<VoicePreview> {
|
|
const response = await this.call('/preview-tts', {
|
|
method: 'POST',
|
|
headers: { 'Content-Type': 'application/json' },
|
|
body: JSON.stringify({
|
|
text,
|
|
tts_voice: options.voice,
|
|
tts_engine: options.engine,
|
|
tts_style: options.style,
|
|
gemini_api_key: options.geminiApiKey,
|
|
}),
|
|
timeoutMs: TTS_TIMEOUT_MS,
|
|
});
|
|
const data = await response.json() as {
|
|
audio_base64: string;
|
|
duration: number;
|
|
words: TimedWord[];
|
|
engine: string;
|
|
voice: string;
|
|
warnings?: string[];
|
|
};
|
|
return {
|
|
audio: Buffer.from(data.audio_base64, 'base64'),
|
|
duration: data.duration,
|
|
words: data.words ?? [],
|
|
engine: data.engine,
|
|
voice: data.voice,
|
|
warnings: data.warnings ?? [],
|
|
};
|
|
}
|
|
}
|
|
|
|
export const ffmpegService = new FFmpegService();
|