Files
Socialflow/server/services/ffmpeg.ts
T
Claude fdace9d8d5 Qwen TTS : serveur GPU distant (Unraid) réglé dans les paramètres
- qwen-tts : GPU NVIDIA détecté automatiquement (bfloat16), image CUDA 12.8
  (RTX 50xx comprises) via TORCH_VARIANT=cu128, docker-compose.gpu.yml
  dédié (clé obligatoire, choix de la carte) et guide d'installation Unraid.
- Paramètres → « Qwen TTS (voix locale) » : adresse + clé du service, test
  de connexion (modèle et carte graphique affichés), stockés en base
  (app_config.qwen_tts_url / qwen_tts_api_key) et transmis à ffmpeg-api à
  chaque voix ; QWEN_TTS_URL reste un défaut.
- Le service qwen-tts CPU du docker-compose principal devient optionnel
  (profil qwen-local).

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01Upu97wMmsRkBoj6iVM4rH6
2026-09-29 09:57:32 +00:00

357 lines
13 KiB
TypeScript

/**
* Client du service FFmpeg (conteneur Python `ffmpeg-service`).
*
* Le service rend le Reel puis expose le MP4 en téléchargement
* (GET /files/{job}/output.mp4) : la vidéo ne transite plus en base64 dans du
* JSON, qui gonflait sa taille d'un tiers et la gardait entière en mémoire.
*/
import fs from 'fs';
import { Readable } from 'stream';
import { pipeline } from 'stream/promises';
import type { ReadableStream as WebReadableStream } from 'stream/web';
import type { SrtCue } from '@shared/srt';
import type { TtsEngine, TtsStyle, VoiceOption } from '@shared/voices';
/** Un rendu long (stabilisation + encodage) peut dépasser plusieurs minutes. */
const PROCESS_TIMEOUT_MS = 20 * 60_000;
const DOWNLOAD_TIMEOUT_MS = 5 * 60_000;
// Qwen (local, CPU) met plusieurs fois la durée de la voix à la générer
const TTS_TIMEOUT_MS = 15 * 60_000;
const HEALTH_TIMEOUT_MS = 5_000;
export interface ReelRenderOptions {
text?: string;
/** Sous-titres SRT : remplacent `text`, la voix lit chacun à son instant. */
srtCues?: SrtCue[];
musicUrl?: string;
ttsEnabled?: boolean;
ttsVoice?: string;
ttsEngine?: TtsEngine;
ttsStyle?: TtsStyle;
geminiApiKey?: string;
qwen?: QwenTarget;
fontSize?: number;
musicVolume?: number;
drawText?: boolean;
stabilize?: boolean;
watermarkUrl?: string;
storeName?: string;
enableEndingEffect?: boolean;
/** Petit logo pendant la vidéo (le grand logo de fin dépend de enableEndingEffect). */
showLogo?: boolean;
}
export interface ReelRenderResult {
video: Buffer;
duration: number;
ttsEngine?: string | null;
ttsVoice?: string | null;
warnings: string[];
}
/** Rendu préparé pour Remotion : fichiers déjà écrits sur le disque. */
export interface PreparedReel {
videoPath: string;
audioPath: string | null;
totalDuration: number;
videoDuration: number;
logoStart: number | null;
/** Fondu final, calculé par le service pour ne jamais rogner la voix. */
fadeStart: number | null;
fadeDuration: number | null;
words: TimedWord[];
ttsEngine?: string | null;
ttsVoice?: string | null;
warnings: string[];
}
export interface TimedWord {
text: string;
start: number;
end: number;
}
/** Service Qwen TTS (local ou serveur GPU) réglé dans les paramètres. */
export interface QwenTarget {
url: string;
apiKey?: string;
}
export interface VoiceCatalog {
qwen: VoiceOption[];
qwen_available: boolean;
/** Modèle et carte graphique du service Qwen, null s'il ne répond pas. */
qwen_status: { model?: string; device?: string; busy?: boolean } | null;
gemini: VoiceOption[];
edge: VoiceOption[];
}
export interface VoicePreview {
audio: Buffer;
duration: number;
words: TimedWord[];
engine: string;
voice: string;
warnings: string[];
}
interface FFmpegConfig {
apiUrl: string;
apiKey: string;
}
/** Erreur renvoyée par le service, avec son message lisible. */
export class FFmpegServiceError extends Error {
constructor(message: string, readonly status?: number) {
super(message);
this.name = 'FFmpegServiceError';
}
}
export class FFmpegService {
private config: FFmpegConfig | null = null;
configure(apiUrl: string, apiKey: string): void {
this.config = { apiUrl: apiUrl.replace(/\/$/, ''), apiKey };
console.log('🎬 FFmpeg Service configured:', this.config.apiUrl);
}
private ensureConfigured(): FFmpegConfig {
if (!this.config) {
throw new FFmpegServiceError("Le service FFmpeg n'est pas configuré (FFMPEG_API_URL / FFMPEG_API_KEY).");
}
return this.config;
}
private async call(path: string, init: RequestInit & { timeoutMs: number }): Promise<Response> {
const config = this.ensureConfigured();
const { timeoutMs, headers, ...rest } = init;
const response = await fetch(`${config.apiUrl}${path}`, {
...rest,
headers: { 'X-API-Key': config.apiKey, ...headers },
signal: AbortSignal.timeout(timeoutMs),
});
if (!response.ok) {
const body = await response.text();
let detail = body;
try {
detail = JSON.parse(body).detail ?? body;
} catch { /* corps non JSON */ }
throw new FFmpegServiceError(String(detail).slice(0, 2000), response.status);
}
return response;
}
/**
* Rend un Reel à partir de l'URL d'une vidéo et renvoie le MP4 produit.
* Lève FFmpegServiceError en cas d'échec (voix comprise).
*/
async renderReel(videoUrl: string, options: ReelRenderOptions = {}): Promise<ReelRenderResult> {
const body = {
video_url: videoUrl,
text: options.text,
srt_cues: options.srtCues,
music_url: options.musicUrl,
tts_enabled: options.ttsEnabled ?? false,
tts_voice: options.ttsVoice,
tts_engine: options.ttsEngine,
tts_style: options.ttsStyle,
gemini_api_key: options.geminiApiKey,
qwen_tts_url: options.qwen?.url,
qwen_tts_api_key: options.qwen?.apiKey,
font_size: options.fontSize ?? 64,
music_volume: options.musicVolume ?? 0.25,
draw_text: options.drawText ?? true,
stabilize: options.stabilize ?? false,
watermark_url: options.watermarkUrl,
store_name: options.storeName,
enable_ending_effect: options.enableEndingEffect ?? true,
show_watermark: options.showLogo ?? true,
};
console.log('🎬 Rendu du Reel :', {
videoUrl,
textLength: options.text?.length ?? 0,
music: !!options.musicUrl,
tts: options.ttsEnabled ? `${options.ttsEngine}/${options.ttsVoice}/${options.ttsStyle ?? 'neutral'}` : false,
});
const response = await this.call('/process-reel', {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify(body),
timeoutMs: PROCESS_TIMEOUT_MS,
});
const data = await response.json() as {
job_id: string;
output_path: string;
duration: number;
tts_engine?: string | null;
tts_voice?: string | null;
warnings?: string[];
};
try {
const file = await this.call(data.output_path, { method: 'GET', timeoutMs: DOWNLOAD_TIMEOUT_MS });
const video = Buffer.from(await file.arrayBuffer());
for (const warning of data.warnings ?? []) console.warn(`⚠️ [FFmpeg] ${warning}`);
return {
video,
duration: data.duration,
ttsEngine: data.tts_engine,
ttsVoice: data.tts_voice,
warnings: data.warnings ?? [],
};
} finally {
// Le fichier n'est plus utile au service une fois récupéré
this.call(`/jobs/${data.job_id}`, { method: 'DELETE', timeoutMs: HEALTH_TIMEOUT_MS })
.catch(() => { /* purgé plus tard par le service */ });
}
}
/**
* Prépare un rendu Remotion : image recadrée à la durée finale et piste son
* finale, écrites dans `targetDir`, plus le minutage des mots.
*/
async prepareReel(
videoUrl: string,
options: ReelRenderOptions & { hasLogo: boolean },
targetDir: string,
): Promise<PreparedReel> {
const response = await this.call('/prepare-reel', {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({
video_url: videoUrl,
text: options.text,
srt_cues: options.srtCues,
music_url: options.musicUrl,
tts_enabled: options.ttsEnabled ?? false,
tts_voice: options.ttsVoice,
tts_engine: options.ttsEngine,
tts_style: options.ttsStyle,
gemini_api_key: options.geminiApiKey,
qwen_tts_url: options.qwen?.url,
qwen_tts_api_key: options.qwen?.apiKey,
music_volume: options.musicVolume ?? 0.25,
draw_text: options.drawText ?? true,
stabilize: options.stabilize ?? false,
store_name: options.storeName,
enable_ending_effect: options.enableEndingEffect ?? true,
has_logo: options.hasLogo,
}),
timeoutMs: PROCESS_TIMEOUT_MS,
});
const data = await response.json() as {
job_id: string;
video_path: string;
audio_path: string | null;
total_duration: number;
video_duration: number;
logo_start: number | null;
fade_start?: number | null;
fade_duration?: number | null;
words: TimedWord[];
tts_engine?: string | null;
tts_voice?: string | null;
warnings?: string[];
};
try {
await fs.promises.mkdir(targetDir, { recursive: true });
const videoPath = `${targetDir}/video.mp4`;
await this.downloadTo(data.video_path, videoPath);
let audioPath: string | null = null;
if (data.audio_path) {
audioPath = `${targetDir}/audio.wav`;
await this.downloadTo(data.audio_path, audioPath);
}
for (const warning of data.warnings ?? []) console.warn(`⚠️ [FFmpeg] ${warning}`);
return {
videoPath,
audioPath,
totalDuration: data.total_duration,
videoDuration: data.video_duration,
logoStart: data.logo_start,
fadeStart: data.fade_start ?? null,
fadeDuration: data.fade_duration ?? null,
words: data.words ?? [],
ttsEngine: data.tts_engine,
ttsVoice: data.tts_voice,
warnings: data.warnings ?? [],
};
} finally {
this.call(`/jobs/${data.job_id}`, { method: 'DELETE', timeoutMs: HEALTH_TIMEOUT_MS })
.catch(() => { /* purgé plus tard par le service */ });
}
}
/** Télécharge un fichier du service en flux, sans le charger en mémoire. */
private async downloadTo(servicePath: string, target: string): Promise<void> {
const response = await this.call(servicePath, { method: 'GET', timeoutMs: DOWNLOAD_TIMEOUT_MS });
if (!response.body) throw new FFmpegServiceError(`Réponse vide pour ${servicePath}`);
await pipeline(
Readable.fromWeb(response.body as unknown as WebReadableStream<Uint8Array>),
fs.createWriteStream(target),
);
}
async healthCheck(): Promise<boolean> {
try {
await this.call('/health', { method: 'GET', timeoutMs: HEALTH_TIMEOUT_MS });
return true;
} catch {
return false;
}
}
/** Voix proposées par moteur ; `qwen_available` indique si le service local répond. */
async listVoices(qwen?: QwenTarget): Promise<VoiceCatalog> {
const headers: Record<string, string> = {};
if (qwen?.url) headers['X-Qwen-Url'] = qwen.url;
if (qwen?.apiKey) headers['X-Qwen-Key'] = qwen.apiKey;
const response = await this.call('/voices', { method: 'GET', headers, timeoutMs: 30_000 });
return await response.json() as VoiceCatalog;
}
/** Génère la voix seule (aperçu), avec le minutage de chaque mot. */
async previewVoice(
text: string,
options: { voice?: string; engine?: TtsEngine; style?: TtsStyle; geminiApiKey?: string; qwen?: QwenTarget } = {},
): Promise<VoicePreview> {
const response = await this.call('/preview-tts', {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({
text,
tts_voice: options.voice,
tts_engine: options.engine,
tts_style: options.style,
gemini_api_key: options.geminiApiKey,
qwen_tts_url: options.qwen?.url,
qwen_tts_api_key: options.qwen?.apiKey,
}),
timeoutMs: TTS_TIMEOUT_MS,
});
const data = await response.json() as {
audio_base64: string;
duration: number;
words: TimedWord[];
engine: string;
voice: string;
warnings?: string[];
};
return {
audio: Buffer.from(data.audio_base64, 'base64'),
duration: data.duration,
words: data.words ?? [],
engine: data.engine,
voice: data.voice,
warnings: data.warnings ?? [],
};
}
}
export const ffmpegService = new FFmpegService();