mirror of
https://github.com/R0m1k3/Socialflow.git
synced 2026-10-11 17:26:45 +02:00
- qwen-tts : GPU NVIDIA détecté automatiquement (bfloat16), image CUDA 12.8 (RTX 50xx comprises) via TORCH_VARIANT=cu128, docker-compose.gpu.yml dédié (clé obligatoire, choix de la carte) et guide d'installation Unraid. - Paramètres → « Qwen TTS (voix locale) » : adresse + clé du service, test de connexion (modèle et carte graphique affichés), stockés en base (app_config.qwen_tts_url / qwen_tts_api_key) et transmis à ffmpeg-api à chaque voix ; QWEN_TTS_URL reste un défaut. - Le service qwen-tts CPU du docker-compose principal devient optionnel (profil qwen-local). Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Upu97wMmsRkBoj6iVM4rH6
357 lines
13 KiB
TypeScript
357 lines
13 KiB
TypeScript
/**
|
|
* Client du service FFmpeg (conteneur Python `ffmpeg-service`).
|
|
*
|
|
* Le service rend le Reel puis expose le MP4 en téléchargement
|
|
* (GET /files/{job}/output.mp4) : la vidéo ne transite plus en base64 dans du
|
|
* JSON, qui gonflait sa taille d'un tiers et la gardait entière en mémoire.
|
|
*/
|
|
|
|
import fs from 'fs';
|
|
import { Readable } from 'stream';
|
|
import { pipeline } from 'stream/promises';
|
|
import type { ReadableStream as WebReadableStream } from 'stream/web';
|
|
import type { SrtCue } from '@shared/srt';
|
|
import type { TtsEngine, TtsStyle, VoiceOption } from '@shared/voices';
|
|
|
|
/** Un rendu long (stabilisation + encodage) peut dépasser plusieurs minutes. */
|
|
const PROCESS_TIMEOUT_MS = 20 * 60_000;
|
|
const DOWNLOAD_TIMEOUT_MS = 5 * 60_000;
|
|
// Qwen (local, CPU) met plusieurs fois la durée de la voix à la générer
|
|
const TTS_TIMEOUT_MS = 15 * 60_000;
|
|
const HEALTH_TIMEOUT_MS = 5_000;
|
|
|
|
export interface ReelRenderOptions {
|
|
text?: string;
|
|
/** Sous-titres SRT : remplacent `text`, la voix lit chacun à son instant. */
|
|
srtCues?: SrtCue[];
|
|
musicUrl?: string;
|
|
ttsEnabled?: boolean;
|
|
ttsVoice?: string;
|
|
ttsEngine?: TtsEngine;
|
|
ttsStyle?: TtsStyle;
|
|
geminiApiKey?: string;
|
|
qwen?: QwenTarget;
|
|
fontSize?: number;
|
|
musicVolume?: number;
|
|
drawText?: boolean;
|
|
stabilize?: boolean;
|
|
watermarkUrl?: string;
|
|
storeName?: string;
|
|
enableEndingEffect?: boolean;
|
|
/** Petit logo pendant la vidéo (le grand logo de fin dépend de enableEndingEffect). */
|
|
showLogo?: boolean;
|
|
}
|
|
|
|
export interface ReelRenderResult {
|
|
video: Buffer;
|
|
duration: number;
|
|
ttsEngine?: string | null;
|
|
ttsVoice?: string | null;
|
|
warnings: string[];
|
|
}
|
|
|
|
/** Rendu préparé pour Remotion : fichiers déjà écrits sur le disque. */
|
|
export interface PreparedReel {
|
|
videoPath: string;
|
|
audioPath: string | null;
|
|
totalDuration: number;
|
|
videoDuration: number;
|
|
logoStart: number | null;
|
|
/** Fondu final, calculé par le service pour ne jamais rogner la voix. */
|
|
fadeStart: number | null;
|
|
fadeDuration: number | null;
|
|
words: TimedWord[];
|
|
ttsEngine?: string | null;
|
|
ttsVoice?: string | null;
|
|
warnings: string[];
|
|
}
|
|
|
|
export interface TimedWord {
|
|
text: string;
|
|
start: number;
|
|
end: number;
|
|
}
|
|
|
|
/** Service Qwen TTS (local ou serveur GPU) réglé dans les paramètres. */
|
|
export interface QwenTarget {
|
|
url: string;
|
|
apiKey?: string;
|
|
}
|
|
|
|
export interface VoiceCatalog {
|
|
qwen: VoiceOption[];
|
|
qwen_available: boolean;
|
|
/** Modèle et carte graphique du service Qwen, null s'il ne répond pas. */
|
|
qwen_status: { model?: string; device?: string; busy?: boolean } | null;
|
|
gemini: VoiceOption[];
|
|
edge: VoiceOption[];
|
|
}
|
|
|
|
export interface VoicePreview {
|
|
audio: Buffer;
|
|
duration: number;
|
|
words: TimedWord[];
|
|
engine: string;
|
|
voice: string;
|
|
warnings: string[];
|
|
}
|
|
|
|
interface FFmpegConfig {
|
|
apiUrl: string;
|
|
apiKey: string;
|
|
}
|
|
|
|
/** Erreur renvoyée par le service, avec son message lisible. */
|
|
export class FFmpegServiceError extends Error {
|
|
constructor(message: string, readonly status?: number) {
|
|
super(message);
|
|
this.name = 'FFmpegServiceError';
|
|
}
|
|
}
|
|
|
|
export class FFmpegService {
|
|
private config: FFmpegConfig | null = null;
|
|
|
|
configure(apiUrl: string, apiKey: string): void {
|
|
this.config = { apiUrl: apiUrl.replace(/\/$/, ''), apiKey };
|
|
console.log('🎬 FFmpeg Service configured:', this.config.apiUrl);
|
|
}
|
|
|
|
private ensureConfigured(): FFmpegConfig {
|
|
if (!this.config) {
|
|
throw new FFmpegServiceError("Le service FFmpeg n'est pas configuré (FFMPEG_API_URL / FFMPEG_API_KEY).");
|
|
}
|
|
return this.config;
|
|
}
|
|
|
|
private async call(path: string, init: RequestInit & { timeoutMs: number }): Promise<Response> {
|
|
const config = this.ensureConfigured();
|
|
const { timeoutMs, headers, ...rest } = init;
|
|
const response = await fetch(`${config.apiUrl}${path}`, {
|
|
...rest,
|
|
headers: { 'X-API-Key': config.apiKey, ...headers },
|
|
signal: AbortSignal.timeout(timeoutMs),
|
|
});
|
|
if (!response.ok) {
|
|
const body = await response.text();
|
|
let detail = body;
|
|
try {
|
|
detail = JSON.parse(body).detail ?? body;
|
|
} catch { /* corps non JSON */ }
|
|
throw new FFmpegServiceError(String(detail).slice(0, 2000), response.status);
|
|
}
|
|
return response;
|
|
}
|
|
|
|
/**
|
|
* Rend un Reel à partir de l'URL d'une vidéo et renvoie le MP4 produit.
|
|
* Lève FFmpegServiceError en cas d'échec (voix comprise).
|
|
*/
|
|
async renderReel(videoUrl: string, options: ReelRenderOptions = {}): Promise<ReelRenderResult> {
|
|
const body = {
|
|
video_url: videoUrl,
|
|
text: options.text,
|
|
srt_cues: options.srtCues,
|
|
music_url: options.musicUrl,
|
|
tts_enabled: options.ttsEnabled ?? false,
|
|
tts_voice: options.ttsVoice,
|
|
tts_engine: options.ttsEngine,
|
|
tts_style: options.ttsStyle,
|
|
gemini_api_key: options.geminiApiKey,
|
|
qwen_tts_url: options.qwen?.url,
|
|
qwen_tts_api_key: options.qwen?.apiKey,
|
|
font_size: options.fontSize ?? 64,
|
|
music_volume: options.musicVolume ?? 0.25,
|
|
draw_text: options.drawText ?? true,
|
|
stabilize: options.stabilize ?? false,
|
|
watermark_url: options.watermarkUrl,
|
|
store_name: options.storeName,
|
|
enable_ending_effect: options.enableEndingEffect ?? true,
|
|
show_watermark: options.showLogo ?? true,
|
|
};
|
|
|
|
console.log('🎬 Rendu du Reel :', {
|
|
videoUrl,
|
|
textLength: options.text?.length ?? 0,
|
|
music: !!options.musicUrl,
|
|
tts: options.ttsEnabled ? `${options.ttsEngine}/${options.ttsVoice}/${options.ttsStyle ?? 'neutral'}` : false,
|
|
});
|
|
|
|
const response = await this.call('/process-reel', {
|
|
method: 'POST',
|
|
headers: { 'Content-Type': 'application/json' },
|
|
body: JSON.stringify(body),
|
|
timeoutMs: PROCESS_TIMEOUT_MS,
|
|
});
|
|
const data = await response.json() as {
|
|
job_id: string;
|
|
output_path: string;
|
|
duration: number;
|
|
tts_engine?: string | null;
|
|
tts_voice?: string | null;
|
|
warnings?: string[];
|
|
};
|
|
|
|
try {
|
|
const file = await this.call(data.output_path, { method: 'GET', timeoutMs: DOWNLOAD_TIMEOUT_MS });
|
|
const video = Buffer.from(await file.arrayBuffer());
|
|
for (const warning of data.warnings ?? []) console.warn(`⚠️ [FFmpeg] ${warning}`);
|
|
return {
|
|
video,
|
|
duration: data.duration,
|
|
ttsEngine: data.tts_engine,
|
|
ttsVoice: data.tts_voice,
|
|
warnings: data.warnings ?? [],
|
|
};
|
|
} finally {
|
|
// Le fichier n'est plus utile au service une fois récupéré
|
|
this.call(`/jobs/${data.job_id}`, { method: 'DELETE', timeoutMs: HEALTH_TIMEOUT_MS })
|
|
.catch(() => { /* purgé plus tard par le service */ });
|
|
}
|
|
}
|
|
|
|
/**
|
|
* Prépare un rendu Remotion : image recadrée à la durée finale et piste son
|
|
* finale, écrites dans `targetDir`, plus le minutage des mots.
|
|
*/
|
|
async prepareReel(
|
|
videoUrl: string,
|
|
options: ReelRenderOptions & { hasLogo: boolean },
|
|
targetDir: string,
|
|
): Promise<PreparedReel> {
|
|
const response = await this.call('/prepare-reel', {
|
|
method: 'POST',
|
|
headers: { 'Content-Type': 'application/json' },
|
|
body: JSON.stringify({
|
|
video_url: videoUrl,
|
|
text: options.text,
|
|
srt_cues: options.srtCues,
|
|
music_url: options.musicUrl,
|
|
tts_enabled: options.ttsEnabled ?? false,
|
|
tts_voice: options.ttsVoice,
|
|
tts_engine: options.ttsEngine,
|
|
tts_style: options.ttsStyle,
|
|
gemini_api_key: options.geminiApiKey,
|
|
qwen_tts_url: options.qwen?.url,
|
|
qwen_tts_api_key: options.qwen?.apiKey,
|
|
music_volume: options.musicVolume ?? 0.25,
|
|
draw_text: options.drawText ?? true,
|
|
stabilize: options.stabilize ?? false,
|
|
store_name: options.storeName,
|
|
enable_ending_effect: options.enableEndingEffect ?? true,
|
|
has_logo: options.hasLogo,
|
|
}),
|
|
timeoutMs: PROCESS_TIMEOUT_MS,
|
|
});
|
|
const data = await response.json() as {
|
|
job_id: string;
|
|
video_path: string;
|
|
audio_path: string | null;
|
|
total_duration: number;
|
|
video_duration: number;
|
|
logo_start: number | null;
|
|
fade_start?: number | null;
|
|
fade_duration?: number | null;
|
|
words: TimedWord[];
|
|
tts_engine?: string | null;
|
|
tts_voice?: string | null;
|
|
warnings?: string[];
|
|
};
|
|
|
|
try {
|
|
await fs.promises.mkdir(targetDir, { recursive: true });
|
|
const videoPath = `${targetDir}/video.mp4`;
|
|
await this.downloadTo(data.video_path, videoPath);
|
|
let audioPath: string | null = null;
|
|
if (data.audio_path) {
|
|
audioPath = `${targetDir}/audio.wav`;
|
|
await this.downloadTo(data.audio_path, audioPath);
|
|
}
|
|
for (const warning of data.warnings ?? []) console.warn(`⚠️ [FFmpeg] ${warning}`);
|
|
return {
|
|
videoPath,
|
|
audioPath,
|
|
totalDuration: data.total_duration,
|
|
videoDuration: data.video_duration,
|
|
logoStart: data.logo_start,
|
|
fadeStart: data.fade_start ?? null,
|
|
fadeDuration: data.fade_duration ?? null,
|
|
words: data.words ?? [],
|
|
ttsEngine: data.tts_engine,
|
|
ttsVoice: data.tts_voice,
|
|
warnings: data.warnings ?? [],
|
|
};
|
|
} finally {
|
|
this.call(`/jobs/${data.job_id}`, { method: 'DELETE', timeoutMs: HEALTH_TIMEOUT_MS })
|
|
.catch(() => { /* purgé plus tard par le service */ });
|
|
}
|
|
}
|
|
|
|
/** Télécharge un fichier du service en flux, sans le charger en mémoire. */
|
|
private async downloadTo(servicePath: string, target: string): Promise<void> {
|
|
const response = await this.call(servicePath, { method: 'GET', timeoutMs: DOWNLOAD_TIMEOUT_MS });
|
|
if (!response.body) throw new FFmpegServiceError(`Réponse vide pour ${servicePath}`);
|
|
await pipeline(
|
|
Readable.fromWeb(response.body as unknown as WebReadableStream<Uint8Array>),
|
|
fs.createWriteStream(target),
|
|
);
|
|
}
|
|
|
|
async healthCheck(): Promise<boolean> {
|
|
try {
|
|
await this.call('/health', { method: 'GET', timeoutMs: HEALTH_TIMEOUT_MS });
|
|
return true;
|
|
} catch {
|
|
return false;
|
|
}
|
|
}
|
|
|
|
/** Voix proposées par moteur ; `qwen_available` indique si le service local répond. */
|
|
async listVoices(qwen?: QwenTarget): Promise<VoiceCatalog> {
|
|
const headers: Record<string, string> = {};
|
|
if (qwen?.url) headers['X-Qwen-Url'] = qwen.url;
|
|
if (qwen?.apiKey) headers['X-Qwen-Key'] = qwen.apiKey;
|
|
const response = await this.call('/voices', { method: 'GET', headers, timeoutMs: 30_000 });
|
|
return await response.json() as VoiceCatalog;
|
|
}
|
|
|
|
/** Génère la voix seule (aperçu), avec le minutage de chaque mot. */
|
|
async previewVoice(
|
|
text: string,
|
|
options: { voice?: string; engine?: TtsEngine; style?: TtsStyle; geminiApiKey?: string; qwen?: QwenTarget } = {},
|
|
): Promise<VoicePreview> {
|
|
const response = await this.call('/preview-tts', {
|
|
method: 'POST',
|
|
headers: { 'Content-Type': 'application/json' },
|
|
body: JSON.stringify({
|
|
text,
|
|
tts_voice: options.voice,
|
|
tts_engine: options.engine,
|
|
tts_style: options.style,
|
|
gemini_api_key: options.geminiApiKey,
|
|
qwen_tts_url: options.qwen?.url,
|
|
qwen_tts_api_key: options.qwen?.apiKey,
|
|
}),
|
|
timeoutMs: TTS_TIMEOUT_MS,
|
|
});
|
|
const data = await response.json() as {
|
|
audio_base64: string;
|
|
duration: number;
|
|
words: TimedWord[];
|
|
engine: string;
|
|
voice: string;
|
|
warnings?: string[];
|
|
};
|
|
return {
|
|
audio: Buffer.from(data.audio_base64, 'base64'),
|
|
duration: data.duration,
|
|
words: data.words ?? [],
|
|
engine: data.engine,
|
|
voice: data.voice,
|
|
warnings: data.warnings ?? [],
|
|
};
|
|
}
|
|
}
|
|
|
|
export const ffmpegService = new FFmpegService();
|