mirror of
https://github.com/R0m1k3/Socialflow.git
synced 2026-10-11 17:26:45 +02:00
Voix : plus d'accélération sur les SRT, moteur local Qwen3-TTS
SRT : la voix n'est plus jamais accélérée (atempo sans plafond, jusqu'à 2× et plus, rendait la voix incompréhensible). Un morceau plus long que son sous-titre déborde et décale les suivants ; les sous-titres suivent la voix, la vidéo s'allonge et un avertissement signale le décalage. Rythme : style par défaut « neutre », consignes Gemini sans « rythme entraînant », débit Edge des styles dynamique/promo ramené à +0 %/+3 %. Nouveau moteur « qwen » : service qwen-tts (Qwen3-TTS sur CPU, gratuit, français, ton piloté par consigne, clonage de voix depuis qwen-tts/voices). Même vérification Whisper que Gemini, généralisée ; repli qwen → gemini → edge, et une lecture sous 85 % du texte est écartée (mots sautés/faux). Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Upu97wMmsRkBoj6iVM4rH6
This commit is contained in:
20 files changed
+579
-98
No files matched your search
@@ -132,6 +132,17 @@ GEMINI_API_KEY=
|
||||
# demande de reconstruire ffmpeg-api.
|
||||
# WHISPER_MODEL=base
|
||||
#
|
||||
# Voix locale Qwen3-TTS (service qwen-tts) : gratuite, française, sans quota,
|
||||
# mais générée sur CPU (5 à 7 fois la durée de la voix sur 4 cœurs).
|
||||
# Modèle choisi à la construction de l'image :
|
||||
# - Qwen/Qwen3-TTS-12Hz-1.7B-CustomVoice (défaut) : ton et émotion pilotés ;
|
||||
# - Qwen/Qwen3-TTS-12Hz-0.6B-CustomVoice : ~30 % plus rapide, moins expressif.
|
||||
# QWEN_TTS_MODEL=Qwen/Qwen3-TTS-12Hz-1.7B-CustomVoice
|
||||
# Clonage des voix déposées dans qwen-tts/voices (vide pour désactiver).
|
||||
# QWEN_TTS_CLONE_MODEL=Qwen/Qwen3-TTS-12Hz-0.6B-Base
|
||||
# QWEN_TTS_THREADS=0
|
||||
# QWEN_TTS_MEMORY=10g
|
||||
#
|
||||
# Moteur de rendu des Reels vidéo :
|
||||
# - remotion (défaut) : sous-titres animés, logo et effet de fin identiques à
|
||||
# l'aperçu affiché dans l'application ; ~30 à 60 s pour un Reel de 30 s ;
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import { useEffect, useRef, useState } from "react";
|
||||
import { useQuery } from "@tanstack/react-query";
|
||||
import { Loader2, Play, Square } from "lucide-react";
|
||||
import { Button } from "@/components/ui/button";
|
||||
import { Label } from "@/components/ui/label";
|
||||
@@ -19,6 +20,7 @@ import {
|
||||
voicesFor,
|
||||
type TtsEngine,
|
||||
type TtsStyle,
|
||||
type VoiceOption,
|
||||
} from "@shared/voices";
|
||||
import type { TimedWord } from "@shared/captions";
|
||||
|
||||
@@ -76,7 +78,13 @@ export function VoicePicker({ value, onChange, sampleText, compact = false, onPr
|
||||
|
||||
useEffect(() => () => audioRef.current?.pause(), []);
|
||||
|
||||
const voices = voicesFor(value.engine);
|
||||
// Voix Qwen réellement installées (voix clonées comprises) ; moteur grisé si le service local ne répond pas
|
||||
const { data: qwenCatalog } = useQuery<{ qwenAvailable: boolean; qwen: VoiceOption[] }>({
|
||||
queryKey: ["/api/reels/voices"],
|
||||
staleTime: 5 * 60_000,
|
||||
});
|
||||
const qwenAvailable = qwenCatalog?.qwenAvailable ?? false;
|
||||
const voices = value.engine === "qwen" && qwenCatalog?.qwen.length ? qwenCatalog.qwen : voicesFor(value.engine);
|
||||
const groups = [
|
||||
{ label: "Voix féminines", items: voices.filter((v) => v.gender === "female") },
|
||||
{ label: "Voix masculines", items: voices.filter((v) => v.gender === "male") },
|
||||
@@ -133,7 +141,18 @@ export function VoicePicker({ value, onChange, sampleText, compact = false, onPr
|
||||
<div className="space-y-3">
|
||||
<div className="flex items-center gap-3">
|
||||
<Label className={`${labelClass} w-14 shrink-0`}>Moteur</Label>
|
||||
<div className="flex gap-2">
|
||||
<div className="flex flex-wrap gap-2">
|
||||
<Button
|
||||
type="button"
|
||||
size="sm"
|
||||
className={compact ? "h-7 text-xs" : undefined}
|
||||
variant={value.engine === "qwen" ? "default" : "outline"}
|
||||
disabled={!qwenAvailable && value.engine !== "qwen"}
|
||||
title={qwenAvailable ? "Voix locale, gratuite, plus lente à générer" : "Service Qwen TTS non démarré"}
|
||||
onClick={() => setEngine("qwen")}
|
||||
>
|
||||
Qwen (locale)
|
||||
</Button>
|
||||
<Button
|
||||
type="button"
|
||||
size="sm"
|
||||
|
||||
@@ -86,11 +86,45 @@ services:
|
||||
environment:
|
||||
API_KEY: ${FFMPEG_API_KEY:-socialflow-secret-ffmpeg-key}
|
||||
GEMINI_TTS_MODEL: ${GEMINI_TTS_MODEL:-gemini-2.5-flash-preview-tts}
|
||||
# Moteur de voix local (service qwen-tts ci-dessous)
|
||||
QWEN_TTS_URL: http://qwen-tts:8001
|
||||
QWEN_TTS_API_KEY: ${FFMPEG_API_KEY:-socialflow-secret-ffmpeg-key}
|
||||
networks:
|
||||
- internal
|
||||
expose:
|
||||
- "8000"
|
||||
|
||||
# Voix locale Qwen3-TTS (gratuite, française, sur CPU). Plus lente que
|
||||
# Gemini : compter 5 à 7 fois la durée de la voix sur 4 cœurs.
|
||||
qwen-tts:
|
||||
build:
|
||||
context: ./qwen-tts
|
||||
dockerfile: Dockerfile
|
||||
args:
|
||||
# 1.7B : ton et émotion pilotés par consigne ; 0.6B : ~30 % plus rapide
|
||||
QWEN_TTS_MODEL: ${QWEN_TTS_MODEL:-Qwen/Qwen3-TTS-12Hz-1.7B-CustomVoice}
|
||||
# Clonage des voix déposées dans qwen-tts/voices (vide : désactivé)
|
||||
QWEN_TTS_CLONE_MODEL: ${QWEN_TTS_CLONE_MODEL-Qwen/Qwen3-TTS-12Hz-0.6B-Base}
|
||||
image: socialflow-qwen-tts:latest
|
||||
pull_policy: build
|
||||
container_name: socialflow-qwen-tts
|
||||
restart: unless-stopped
|
||||
environment:
|
||||
API_KEY: ${FFMPEG_API_KEY:-socialflow-secret-ffmpeg-key}
|
||||
# Threads de calcul (0 : tous les cœurs)
|
||||
QWEN_TTS_THREADS: ${QWEN_TTS_THREADS:-0}
|
||||
volumes:
|
||||
# Voix clonées ajoutables sans reconstruire l'image
|
||||
- ./qwen-tts/voices:/app/voices:ro
|
||||
networks:
|
||||
- internal
|
||||
expose:
|
||||
- "8001"
|
||||
deploy:
|
||||
resources:
|
||||
limits:
|
||||
memory: ${QWEN_TTS_MEMORY:-10g}
|
||||
|
||||
volumes:
|
||||
postgres_data:
|
||||
driver: local
|
||||
|
||||
@@ -100,7 +100,7 @@ async def health():
|
||||
|
||||
@app.get("/voices", dependencies=[Depends(require_key)])
|
||||
async def list_voices():
|
||||
return voices.catalog()
|
||||
return voices.catalog(await tts.qwen.list_voices())
|
||||
|
||||
|
||||
@app.post("/preview-tts", dependencies=[Depends(require_key)])
|
||||
|
||||
@@ -36,3 +36,10 @@ FPS = 30
|
||||
VOICE_DELAY = 2.0
|
||||
|
||||
SUBTITLE_FONT = os.environ.get("SUBTITLE_FONT", "Montserrat")
|
||||
|
||||
# Moteur local Qwen3-TTS (service qwen-tts). Vide : moteur indisponible.
|
||||
QWEN_TTS_URL = os.environ.get("QWEN_TTS_URL", "").rstrip("/")
|
||||
# Sur CPU, la génération prend plusieurs fois la durée de la voix
|
||||
QWEN_TTS_TIMEOUT = float(os.environ.get("QWEN_TTS_TIMEOUT", "900"))
|
||||
QWEN_TTS_ATTEMPTS = int(os.environ.get("QWEN_TTS_ATTEMPTS", "2"))
|
||||
QWEN_TTS_API_KEY = os.environ.get("QWEN_TTS_API_KEY", "")
|
||||
@@ -11,8 +11,10 @@ Si la lecture unique ne peut pas être répartie entre les sous-titres, on lit
|
||||
chaque sous-titre séparément (même voix, même graine), en découpant là aussi
|
||||
au niveau sonore mesuré.
|
||||
|
||||
Quand un morceau est plus long que son sous-titre, il est accéléré (sans
|
||||
changer la hauteur de la voix) pour finir à temps.
|
||||
La voix n'est jamais accélérée : quand un morceau est plus long que son
|
||||
sous-titre, il déborde et les morceaux suivants sont décalés d'autant (le
|
||||
minutage SRT n'est qu'un instant de départ au plus tôt). Les sous-titres
|
||||
affichés suivent la voix et la vidéo s'allonge si besoin.
|
||||
"""
|
||||
|
||||
import logging
|
||||
@@ -26,10 +28,10 @@ from .text import clean_text
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
# Débordement toléré sur le silence qui suit un sous-titre avant d'accélérer
|
||||
OVERFLOW_TOLERANCE = 0.25
|
||||
# Au-delà, l'accélération s'entend nettement : l'utilisateur est prévenu
|
||||
AUDIBLE_SPEEDUP = 1.35
|
||||
# Silence ajouté entre deux morceaux décalés (en plus des marges de chaque morceau)
|
||||
MIN_GAP = 0.05
|
||||
# Au-delà de ce retard cumulé sur le minutage SRT, l'utilisateur est prévenu
|
||||
DRIFT_WARNING = 1.0
|
||||
|
||||
_SENTENCE_END = re.compile(r"[.!?…]$")
|
||||
|
||||
@@ -50,31 +52,16 @@ class Segment:
|
||||
words: list[Word]
|
||||
|
||||
|
||||
def cue_windows(cues: list[Cue]) -> list[float]:
|
||||
"""Durée disponible pour lire chaque sous-titre sans empiéter sur le suivant."""
|
||||
windows = []
|
||||
for i, cue in enumerate(cues):
|
||||
limit = cue.end + OVERFLOW_TOLERANCE
|
||||
if i + 1 < len(cues):
|
||||
limit = min(limit, max(cue.end, cues[i + 1].start))
|
||||
windows.append(max(0.1, limit - cue.start))
|
||||
return windows
|
||||
|
||||
|
||||
def speed_factor(duration: float, window: float) -> float:
|
||||
"""Accélération nécessaire pour tenir dans la fenêtre (1 = vitesse normale)."""
|
||||
return max(1.0, duration / window) if window > 0 else 1.0
|
||||
|
||||
|
||||
def atempo_chain(factor: float) -> str:
|
||||
"""Filtre atempo ; découpé en étapes ≤ 2 pour les anciennes versions de FFmpeg."""
|
||||
steps = []
|
||||
remaining = factor
|
||||
while remaining > 2.0:
|
||||
steps.append(2.0)
|
||||
remaining /= 2.0
|
||||
steps.append(remaining)
|
||||
return ",".join(f"atempo={s:.4f}" for s in steps)
|
||||
def place_segments(cues: list[Cue], durations: list[float]) -> list[float]:
|
||||
"""Instant de départ de chaque morceau : celui de son sous-titre, ou juste
|
||||
après le morceau précédent s'il déborde (vitesse naturelle conservée)."""
|
||||
starts: list[float] = []
|
||||
previous_end = 0.0
|
||||
for cue, duration in zip(cues, durations):
|
||||
start = cue.start if not starts else max(cue.start, previous_end + MIN_GAP)
|
||||
starts.append(start)
|
||||
previous_end = start + duration
|
||||
return starts
|
||||
|
||||
|
||||
def as_sentence(text: str) -> str:
|
||||
@@ -162,26 +149,30 @@ async def synthesize_cues(
|
||||
span = cue_envelope.speech_span(0.0, cue_envelope.duration)
|
||||
sources.append((track.path, Segment(span[0], span[1], track.words)))
|
||||
|
||||
durations = [segment.source_end - segment.source_start for _, segment in sources]
|
||||
starts = place_segments(cues, durations)
|
||||
placed: list[tuple[Path, float]] = []
|
||||
words: list[Word] = []
|
||||
for index, (cue, window, (source, segment)) in enumerate(zip(cues, cue_windows(cues), sources)):
|
||||
factor = speed_factor(segment.source_end - segment.source_start, window)
|
||||
if factor > AUDIBLE_SPEEDUP:
|
||||
warnings.append(
|
||||
f"Sous-titre {index + 1} (« {cue.text[:40]} ») lu {factor:.1f}× plus vite "
|
||||
"pour tenir dans son minutage : raccourcissez-le ou allongez sa durée."
|
||||
)
|
||||
for index, (cue, start, (source, segment)) in enumerate(zip(cues, starts, sources)):
|
||||
fitted = workdir / f"segment_{index:03d}.wav"
|
||||
await proc.run(extract_command(source, segment, factor, fitted), timeout=120)
|
||||
placed.append((fitted, cue.start))
|
||||
await proc.run(extract_command(source, segment, fitted), timeout=120)
|
||||
placed.append((fitted, start))
|
||||
added_period = not _SENTENCE_END.search(clean_text(cue.text))
|
||||
for position, w in enumerate(segment.words):
|
||||
text = w.text
|
||||
if added_period and position == len(segment.words) - 1:
|
||||
text = text.rstrip(".") # point ajouté pour la lecture, pas à afficher
|
||||
start = cue.start + max(0.0, w.start - segment.source_start) / factor
|
||||
end = cue.start + max(0.0, w.end - segment.source_start) / factor
|
||||
words.append(Word(text or w.text, start, min(max(end, start + 0.05), cue.start + window)))
|
||||
word_start = start + max(0.0, w.start - segment.source_start)
|
||||
word_end = start + max(0.0, w.end - segment.source_start)
|
||||
words.append(Word(text or w.text, word_start, max(word_end, word_start + 0.05)))
|
||||
|
||||
drift = max((s - c.start for s, c in zip(starts, cues)), default=0.0)
|
||||
if drift > DRIFT_WARNING:
|
||||
warnings.append(
|
||||
f"Le texte lu dépasse le minutage du SRT : les derniers sous-titres sont décalés "
|
||||
f"de {drift:.1f} s (voix gardée à vitesse normale). Raccourcissez le texte "
|
||||
"ou allongez les durées pour rester calé."
|
||||
)
|
||||
|
||||
output = workdir / "voice_srt.wav"
|
||||
await proc.run(mix_command(placed, output), timeout=300)
|
||||
@@ -197,14 +188,14 @@ async def synthesize_cues(
|
||||
return tts.VoiceTrack(output, duration, _monotonic(words), used_engine, used_voice, warnings)
|
||||
|
||||
|
||||
def extract_command(source: Path, segment: Segment, factor: float, output: Path) -> list[str]:
|
||||
"""Extrait un morceau de voix, avec un fondu de 10 ms aux bords (pas de clic)
|
||||
et l'accélération éventuelle."""
|
||||
def extract_command(source: Path, segment: Segment, output: Path) -> list[str]:
|
||||
"""Extrait un morceau de voix, à vitesse normale, avec un fondu de 10 ms
|
||||
aux bords (pas de clic)."""
|
||||
length = segment.source_end - segment.source_start
|
||||
fade_out = max(0.0, length - 0.01)
|
||||
filters = (
|
||||
f"atrim=start={segment.source_start:.3f}:end={segment.source_end:.3f},asetpts=PTS-STARTPTS,"
|
||||
f"afade=t=in:d=0.01,afade=t=out:st={fade_out:.3f}:d=0.01,{atempo_chain(factor)}"
|
||||
f"afade=t=in:d=0.01,afade=t=out:st={fade_out:.3f}:d=0.01"
|
||||
)
|
||||
return [
|
||||
"ffmpeg", "-y", "-hide_banner", "-loglevel", "error", "-i", str(source),
|
||||
|
||||
@@ -1,12 +1,14 @@
|
||||
"""Synthèse vocale : moteur au choix, repli sur Edge, voix traitée et mots minutés."""
|
||||
"""Synthèse vocale : moteur au choix (Qwen local, Gemini, Edge), replis
|
||||
successifs, voix traitée et mots minutés."""
|
||||
|
||||
import logging
|
||||
from collections.abc import Awaitable, Callable
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
|
||||
from .. import align, audio, config, proc, quality
|
||||
from ..align import Word
|
||||
from . import edge, gemini
|
||||
from . import edge, gemini, qwen
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
@@ -35,12 +37,37 @@ async def synthesize(
|
||||
processed: Path | None = None
|
||||
spoken: list[Word] = []
|
||||
used_engine, used_voice = "edge", ""
|
||||
engine = engine or "gemini"
|
||||
|
||||
if (engine or "gemini") == "gemini":
|
||||
if engine == "qwen":
|
||||
result = await _checked(
|
||||
"Qwen",
|
||||
lambda attempt: qwen.synthesize(text, voice, style, workdir, attempt),
|
||||
config.QWEN_TTS_ATTEMPTS,
|
||||
text,
|
||||
workdir,
|
||||
warnings,
|
||||
)
|
||||
if result:
|
||||
processed, spoken, used_voice, reading_ok = result
|
||||
used_engine = "qwen"
|
||||
if reading_ok:
|
||||
processed, spoken = await _trim_stray_sounds(processed, spoken, workdir)
|
||||
elif gemini_api_key:
|
||||
engine = "gemini" # repli sur Gemini quand une clé est disponible
|
||||
|
||||
if engine == "gemini" and processed is None:
|
||||
if not gemini_api_key:
|
||||
warnings.append("Clé Gemini absente : voix Edge utilisée à la place.")
|
||||
else:
|
||||
result = await _gemini_checked(text, voice, style, gemini_api_key, workdir, warnings)
|
||||
result = await _checked(
|
||||
"Gemini",
|
||||
lambda attempt: gemini.synthesize(text, voice, style, gemini_api_key, workdir, attempt),
|
||||
config.GEMINI_TTS_ATTEMPTS,
|
||||
text,
|
||||
workdir,
|
||||
warnings,
|
||||
)
|
||||
if result:
|
||||
processed, spoken, used_voice, reading_ok = result
|
||||
used_engine = "gemini"
|
||||
@@ -61,45 +88,55 @@ async def synthesize(
|
||||
return VoiceTrack(processed, duration, words, used_engine, used_voice, warnings)
|
||||
|
||||
|
||||
async def _gemini_checked(
|
||||
text: str, voice: str | None, style: str | None, api_key: str, workdir: Path, warnings: list[str]
|
||||
# En dessous, une lecture imparfaite est écartée au profit du moteur suivant :
|
||||
# mieux vaut une autre voix qu'une phrase aux mots sautés ou inventés.
|
||||
KEEP_IMPERFECT_COVERAGE = 0.85
|
||||
|
||||
|
||||
async def _checked(
|
||||
name: str,
|
||||
generate: Callable[[int], Awaitable[tuple[Path, str]]],
|
||||
attempts: int,
|
||||
text: str,
|
||||
workdir: Path,
|
||||
warnings: list[str],
|
||||
) -> tuple[Path, list[Word], str, bool] | None:
|
||||
"""Voix Gemini dont la lecture a été vérifiée : Whisper réécoute chaque
|
||||
"""Voix dont la lecture a été vérifiée : Whisper réécoute chaque
|
||||
génération ; mots sautés, fin tronquée ou consigne lue à voix haute
|
||||
déclenchent une nouvelle génération. Renvoie None pour se replier sur Edge."""
|
||||
déclenchent une nouvelle génération. Renvoie None pour passer au moteur suivant."""
|
||||
best: tuple[quality.ReadingCheck, Path, list[Word], str] | None = None
|
||||
last_error: Exception | None = None
|
||||
|
||||
for attempt in range(max(1, config.GEMINI_TTS_ATTEMPTS)):
|
||||
for attempt in range(max(1, attempts)):
|
||||
try:
|
||||
raw, gemini_voice = await gemini.synthesize(text, voice, style, api_key, workdir, attempt)
|
||||
except Exception as error: # noqa: BLE001 — nouvelle tentative, puis repli sur Edge
|
||||
log.warning("Gemini TTS en échec (tentative %d) : %s", attempt + 1, error)
|
||||
raw, engine_voice = await generate(attempt)
|
||||
except Exception as error: # noqa: BLE001 — nouvelle tentative, puis moteur suivant
|
||||
log.warning("%s TTS en échec (tentative %d) : %s", name, attempt + 1, error)
|
||||
last_error = error
|
||||
continue
|
||||
|
||||
processed = workdir / f"voice_gemini_{attempt}.wav"
|
||||
processed = workdir / f"voice_{name.lower()}_{attempt}.wav"
|
||||
await audio.process_voice(raw, processed)
|
||||
spoken = await align.transcribe(processed)
|
||||
check = quality.check_reading(text, spoken)
|
||||
log.info("Lecture Gemini (tentative %d) : %s", attempt + 1, check.describe())
|
||||
log.info("Lecture %s (tentative %d) : %s", name, attempt + 1, check.describe())
|
||||
|
||||
if best is None or _score(check) > _score(best[0]):
|
||||
best = (check, processed, spoken, gemini_voice)
|
||||
best = (check, processed, spoken, engine_voice)
|
||||
if check.acceptable:
|
||||
break
|
||||
|
||||
if best is None:
|
||||
warnings.append(f"Gemini indisponible ({last_error}) : voix Edge utilisée à la place.")
|
||||
warnings.append(f"{name} indisponible ({last_error}) : voix de secours utilisée.")
|
||||
return None
|
||||
|
||||
check, processed, spoken, gemini_voice = best
|
||||
check, processed, spoken, engine_voice = best
|
||||
if not check.acceptable:
|
||||
if check.coverage < 0.7:
|
||||
warnings.append(f"Gemini a mal lu le texte ({check.describe()}) : voix Edge utilisée à la place.")
|
||||
if check.coverage < KEEP_IMPERFECT_COVERAGE:
|
||||
warnings.append(f"{name} a mal lu le texte ({check.describe()}) : voix de secours utilisée.")
|
||||
return None
|
||||
warnings.append(f"Lecture Gemini imparfaite ({check.describe()}) : écoutez le résultat.")
|
||||
return processed, spoken, gemini_voice, check.acceptable
|
||||
warnings.append(f"Lecture {name} imparfaite ({check.describe()}) : écoutez le résultat.")
|
||||
return processed, spoken, engine_voice, check.acceptable
|
||||
|
||||
|
||||
# Au-delà de la fin du dernier mot, ce n'est plus de la parole
|
||||
@@ -108,7 +145,7 @@ STRAY_BEFORE_FIRST_WORD = 0.4
|
||||
|
||||
|
||||
async def _trim_stray_sounds(path: Path, spoken: list[Word], workdir: Path) -> tuple[Path, list[Word]]:
|
||||
"""Retire les bruits que Gemini ajoute parfois avant le premier ou après le
|
||||
"""Retire les bruits que le modèle ajoute parfois avant le premier ou après le
|
||||
dernier mot. Appelé seulement quand la lecture a été vérifiée : le dernier
|
||||
mot est bien reconnu, on ne coupe donc jamais de parole."""
|
||||
if not spoken:
|
||||
|
||||
@@ -13,8 +13,9 @@ log = logging.getLogger(__name__)
|
||||
|
||||
# Réglages de lecture par style (Edge n'interprète pas de consigne en texte)
|
||||
STYLE_PROSODY = {
|
||||
"dynamic": ("+8%", "+2Hz"),
|
||||
"promo": ("+10%", "+3Hz"),
|
||||
# Débit à peine relevé : au-delà, la voix devient difficile à suivre
|
||||
"dynamic": ("+0%", "+2Hz"),
|
||||
"promo": ("+3%", "+3Hz"),
|
||||
"calm": ("-8%", "-2Hz"),
|
||||
"warm": ("-2%", "+0Hz"),
|
||||
}
|
||||
|
||||
@@ -0,0 +1,56 @@
|
||||
"""Synthèse vocale locale Qwen3-TTS (service qwen-tts, sans clé ni quota)."""
|
||||
|
||||
import logging
|
||||
from pathlib import Path
|
||||
|
||||
import httpx
|
||||
|
||||
from .. import config
|
||||
from ..voices import qwen_instruction, resolve_qwen_voice
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class QwenError(RuntimeError):
|
||||
pass
|
||||
|
||||
|
||||
def _headers() -> dict[str, str]:
|
||||
return {"x-api-key": config.QWEN_TTS_API_KEY} if config.QWEN_TTS_API_KEY else {}
|
||||
|
||||
|
||||
async def synthesize(
|
||||
text: str, voice: str | None, style: str | None, workdir: Path, attempt: int = 0
|
||||
) -> tuple[Path, str]:
|
||||
"""Génère la voix ; renvoie un WAV brut et la voix utilisée."""
|
||||
if not config.QWEN_TTS_URL:
|
||||
raise QwenError("service Qwen TTS non configuré (QWEN_TTS_URL)")
|
||||
qwen_voice = resolve_qwen_voice(voice)
|
||||
payload = {"text": text, "voice": qwen_voice, "instruct": qwen_instruction(style), "language": "French"}
|
||||
log.info("Qwen TTS : voix=%s style=%s (%d car.)", qwen_voice, style, len(text))
|
||||
try:
|
||||
async with httpx.AsyncClient(timeout=config.QWEN_TTS_TIMEOUT) as client:
|
||||
response = await client.post(
|
||||
f"{config.QWEN_TTS_URL}/synthesize", json=payload, headers=_headers()
|
||||
)
|
||||
except httpx.HTTPError as error:
|
||||
raise QwenError(f"service Qwen TTS injoignable : {error!r}") from error
|
||||
if response.status_code != 200:
|
||||
raise QwenError(f"Qwen TTS : HTTP {response.status_code} {response.text[:300]}")
|
||||
raw = workdir / f"qwen_raw_{attempt}.wav"
|
||||
raw.write_bytes(response.content)
|
||||
return raw, response.headers.get("x-voice", qwen_voice)
|
||||
|
||||
|
||||
async def list_voices() -> list[dict] | None:
|
||||
"""Voix proposées par le service (clonées comprises) ; None s'il ne répond pas."""
|
||||
if not config.QWEN_TTS_URL:
|
||||
return None
|
||||
try:
|
||||
async with httpx.AsyncClient(timeout=5) as client:
|
||||
response = await client.get(f"{config.QWEN_TTS_URL}/voices", headers=_headers())
|
||||
response.raise_for_status()
|
||||
return response.json()["voices"]
|
||||
except Exception as error: # noqa: BLE001 — moteur simplement signalé indisponible
|
||||
log.warning("Service Qwen TTS indisponible : %s", error)
|
||||
return None
|
||||
@@ -52,13 +52,27 @@ EDGE_VOICES: tuple[Voice, ...] = (
|
||||
Voice("fr-FR-HenriNeural", "Henri", "male"),
|
||||
)
|
||||
|
||||
# Voix prédéfinies de Qwen3-TTS (service local). Les voix clonées
|
||||
# (« clone:nom ») sont listées par le service lui-même.
|
||||
QWEN_VOICES: tuple[Voice, ...] = (
|
||||
Voice("serena", "Serena — douce et chaleureuse", "female"),
|
||||
Voice("vivian", "Vivian — vive et lumineuse", "female"),
|
||||
Voice("sohee", "Sohee — expressive", "female"),
|
||||
Voice("ono_anna", "Anna — espiègle", "female"),
|
||||
Voice("aiden", "Aiden — solaire", "male"),
|
||||
Voice("ryan", "Ryan — dynamique", "male"),
|
||||
Voice("uncle_fu", "Fu — grave et posée", "male"),
|
||||
Voice("dylan", "Dylan — jeune et naturelle", "male"),
|
||||
Voice("eric", "Eric — légèrement voilée", "male"),
|
||||
)
|
||||
|
||||
# Consignes de lecture ajoutées au texte envoyé à Gemini.
|
||||
STYLES: dict[str, tuple[str, str]] = {
|
||||
"neutral": ("Neutre", ""),
|
||||
"dynamic": (
|
||||
"Dynamique",
|
||||
"Lis ce texte en français sur un ton dynamique et enthousiaste, avec un rythme "
|
||||
"entraînant, comme une vidéo courte sur les réseaux sociaux",
|
||||
"Lis ce texte en français sur un ton dynamique et enthousiaste, comme une vidéo "
|
||||
"courte sur les réseaux sociaux, à un débit naturel, sans accélérer, en articulant bien",
|
||||
),
|
||||
"warm": (
|
||||
"Chaleureux",
|
||||
@@ -71,12 +85,25 @@ STYLES: dict[str, tuple[str, str]] = {
|
||||
"promo": (
|
||||
"Promo",
|
||||
"Lis ce texte en français comme une annonce promotionnelle énergique, en "
|
||||
"insistant sur les offres et les prix",
|
||||
"insistant sur les offres et les prix, sans accélérer, en articulant bien",
|
||||
),
|
||||
}
|
||||
|
||||
_GEMINI_BY_ID = {v.id.lower(): v for v in GEMINI_VOICES}
|
||||
_EDGE_BY_ID = {v.id: v for v in EDGE_VOICES}
|
||||
_QWEN_BY_ID = {v.id: v for v in QWEN_VOICES}
|
||||
|
||||
# Consignes de ton pour Qwen (le modèle comprend l'anglais mieux que le français)
|
||||
QWEN_STYLES: dict[str, str] = {
|
||||
"neutral": "Speak natural, native French with clear articulation and a relaxed, natural pace.",
|
||||
"dynamic": "Speak native French in an upbeat, enthusiastic and smiling tone, like a social media video, "
|
||||
"at a natural pace without rushing, clearly articulated.",
|
||||
"warm": "Speak native French in a warm, friendly and caring tone, as if advising a friend, "
|
||||
"at a relaxed pace.",
|
||||
"calm": "Speak native French in a calm, soothing and reassuring tone, slowly and clearly.",
|
||||
"promo": "Speak native French like an energetic promotional announcement, emphasizing offers and prices, "
|
||||
"at a natural pace without rushing, clearly articulated.",
|
||||
}
|
||||
|
||||
|
||||
def resolve_gemini_voice(voice: str | None) -> Voice:
|
||||
@@ -87,7 +114,7 @@ def resolve_gemini_voice(voice: str | None) -> Voice:
|
||||
"""
|
||||
if voice and voice.lower() in _GEMINI_BY_ID:
|
||||
return _GEMINI_BY_ID[voice.lower()]
|
||||
if voice and voice.endswith(("-B", "-D")) or voice == "male":
|
||||
if voice and voice.endswith(("-B", "-D")) or voice_gender(voice) == "male":
|
||||
return _GEMINI_BY_ID["charon"]
|
||||
return _GEMINI_BY_ID["kore"]
|
||||
|
||||
@@ -100,6 +127,8 @@ def voice_gender(voice: str | None) -> str:
|
||||
return _GEMINI_BY_ID[voice.lower()].gender
|
||||
if voice in _EDGE_BY_ID:
|
||||
return _EDGE_BY_ID[voice].gender
|
||||
if voice.lower() in _QWEN_BY_ID:
|
||||
return _QWEN_BY_ID[voice.lower()].gender
|
||||
if voice == "male" or voice.endswith(("-B", "-D")):
|
||||
return "male"
|
||||
return "male" if any(name in voice for name in ("Remy", "Henri", "Paul")) else "female"
|
||||
@@ -113,6 +142,20 @@ def resolve_edge_voice(voice: str | None) -> Voice:
|
||||
return next(v for v in EDGE_VOICES if v.gender == gender)
|
||||
|
||||
|
||||
def resolve_qwen_voice(voice: str | None) -> str:
|
||||
"""Voix Qwen à utiliser : voix clonée ou prédéfinie ; une voix d'un autre
|
||||
moteur est remplacée par une voix Qwen du même genre."""
|
||||
if voice and voice.startswith("clone:"):
|
||||
return voice
|
||||
if voice and voice.lower() in _QWEN_BY_ID:
|
||||
return voice.lower()
|
||||
return "aiden" if voice_gender(voice) == "male" else "serena"
|
||||
|
||||
|
||||
def qwen_instruction(style: str | None) -> str:
|
||||
return QWEN_STYLES.get(style or "neutral", QWEN_STYLES["neutral"])
|
||||
|
||||
|
||||
def edge_fallbacks(primary: Voice) -> list[Voice]:
|
||||
"""Voix françaises de secours du même genre (jamais d'anglais)."""
|
||||
same = [v for v in EDGE_VOICES if v.gender == primary.gender and v != primary]
|
||||
@@ -123,8 +166,11 @@ def style_instruction(style: str | None) -> str:
|
||||
return STYLES.get(style or "neutral", STYLES["neutral"])[1]
|
||||
|
||||
|
||||
def catalog() -> dict:
|
||||
def catalog(qwen_voices: list[dict] | None = None) -> dict:
|
||||
"""Voix par moteur ; `qwen_voices` vient du service Qwen (None s'il ne répond pas)."""
|
||||
return {
|
||||
"qwen": qwen_voices or [v.__dict__ for v in QWEN_VOICES],
|
||||
"qwen_available": qwen_voices is not None,
|
||||
"gemini": [v.__dict__ for v in GEMINI_VOICES],
|
||||
"edge": [v.__dict__ for v in EDGE_VOICES],
|
||||
"styles": [{"id": k, "label": v[0]} for k, v in STYLES.items()],
|
||||
|
||||
@@ -7,29 +7,32 @@ from app.quality import Envelope
|
||||
from app.render import RenderPlan
|
||||
from app.srt_voice import (
|
||||
Cue,
|
||||
Segment,
|
||||
as_sentence,
|
||||
atempo_chain,
|
||||
cue_windows,
|
||||
extract_command,
|
||||
mix_command,
|
||||
place_segments,
|
||||
segments_from_reading,
|
||||
speed_factor,
|
||||
split_by_cue,
|
||||
)
|
||||
|
||||
|
||||
def test_window_stops_at_next_cue_and_tolerates_a_short_overflow():
|
||||
cues = [Cue(0.0, 2.0, "a"), Cue(2.1, 4.0, "b"), Cue(6.0, 7.0, "c")]
|
||||
assert cue_windows(cues) == [2.1, 2.15, 1.25]
|
||||
def test_segments_start_at_their_cue_when_they_fit():
|
||||
cues = [Cue(0.0, 2.0, "a"), Cue(2.5, 4.0, "b"), Cue(6.0, 7.0, "c")]
|
||||
assert place_segments(cues, [2.0, 1.0, 1.0]) == [0.0, 2.5, 6.0]
|
||||
|
||||
|
||||
def test_voice_is_sped_up_only_when_too_long():
|
||||
assert speed_factor(1.5, 2.0) == 1.0
|
||||
assert speed_factor(3.0, 2.0) == 1.5
|
||||
def test_overflowing_segment_pushes_the_next_ones_instead_of_speeding_up():
|
||||
cues = [Cue(0.0, 1.0, "a"), Cue(1.0, 2.0, "b"), Cue(5.0, 6.0, "c")]
|
||||
starts = place_segments(cues, [2.0, 2.0, 1.0])
|
||||
assert starts[0] == 0.0
|
||||
assert abs(starts[1] - 2.05) < 1e-9 # après la fin du premier, à vitesse normale
|
||||
assert starts[2] == 5.0 # rattrape le minutage dès que possible
|
||||
|
||||
|
||||
def test_atempo_is_split_into_supported_steps():
|
||||
assert atempo_chain(1.25) == "atempo=1.2500"
|
||||
assert atempo_chain(3.0) == "atempo=2.0000,atempo=1.5000"
|
||||
def test_segments_are_never_sped_up():
|
||||
cmd = extract_command(Path("in.wav"), Segment(1.0, 4.0, []), Path("out.wav"))
|
||||
assert "atempo" not in cmd[cmd.index("-af") + 1]
|
||||
|
||||
|
||||
def test_each_cue_becomes_a_sentence_for_the_reading():
|
||||
|
||||
@@ -21,3 +21,18 @@ def test_edge_fallbacks_are_french_and_same_gender():
|
||||
def test_style_prompt_precedes_text():
|
||||
assert build_prompt("Salut", "neutral") == "Salut"
|
||||
assert build_prompt("Salut", "dynamic").endswith(":\nSalut")
|
||||
|
||||
|
||||
def test_qwen_voice_keeps_clones_and_matches_gender():
|
||||
from app.voices import qwen_instruction, resolve_qwen_voice
|
||||
|
||||
assert resolve_qwen_voice("clone:camille") == "clone:camille"
|
||||
assert resolve_qwen_voice("Ryan") == "ryan"
|
||||
assert resolve_qwen_voice("Charon") == "aiden" # voix Gemini masculine
|
||||
assert resolve_qwen_voice(None) == "serena"
|
||||
assert "French" in qwen_instruction("dynamic")
|
||||
|
||||
|
||||
def test_qwen_voice_falls_back_to_same_gender_elsewhere():
|
||||
assert resolve_edge_voice("uncle_fu").gender == "male"
|
||||
assert resolve_gemini_voice("uncle_fu").id == "Charon"
|
||||
@@ -0,0 +1,33 @@
|
||||
FROM python:3.11-slim
|
||||
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
libsndfile1 sox ffmpeg \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
# PyTorch CPU (pas de CUDA : image bien plus légère). torch et torchaudio
|
||||
# doivent avoir la même version, sinon torchaudio ne se charge pas.
|
||||
RUN pip install --no-cache-dir torch==2.8.0 torchaudio==2.8.0 \
|
||||
--index-url https://download.pytorch.org/whl/cpu
|
||||
COPY requirements.txt .
|
||||
RUN pip install --no-cache-dir -r requirements.txt
|
||||
|
||||
# Modèles téléchargés au build : aucun accès réseau au premier rendu.
|
||||
# CustomVoice : voix prédéfinies pilotées par une consigne (ton, émotion).
|
||||
# Base : clonage d'une voix de référence (voices/*.wav), chargé seulement s'il y en a.
|
||||
ARG QWEN_TTS_MODEL=Qwen/Qwen3-TTS-12Hz-0.6B-CustomVoice
|
||||
ARG QWEN_TTS_CLONE_MODEL=Qwen/Qwen3-TTS-12Hz-0.6B-Base
|
||||
ENV QWEN_TTS_MODEL=${QWEN_TTS_MODEL} \
|
||||
QWEN_TTS_CLONE_MODEL=${QWEN_TTS_CLONE_MODEL} \
|
||||
HF_HOME=/opt/hf-cache \
|
||||
HOME=/tmp
|
||||
RUN python -c "from huggingface_hub import snapshot_download as d; d('${QWEN_TTS_MODEL}'); d('${QWEN_TTS_CLONE_MODEL}') if '${QWEN_TTS_CLONE_MODEL}' else None"
|
||||
ENV HF_HUB_OFFLINE=1
|
||||
|
||||
COPY main.py .
|
||||
COPY voices ./voices
|
||||
|
||||
EXPOSE 8001
|
||||
|
||||
CMD ["uvicorn", "main:app", "--host", "0.0.0.0", "--port", "8001"]
|
||||
@@ -0,0 +1,159 @@
|
||||
"""Synthèse vocale locale Qwen3-TTS (CPU), appelée par le service ffmpeg-api.
|
||||
|
||||
- Voix prédéfinies (modèle CustomVoice) : le ton est donné par une consigne.
|
||||
- Voix clonées : un extrait de référence dans voices/ (modèle Base, chargé
|
||||
seulement si au moins une voix y est déposée).
|
||||
|
||||
Une seule génération à la fois : chacune occupe tous les cœurs.
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import contextlib
|
||||
import io
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import time
|
||||
from pathlib import Path
|
||||
|
||||
import soundfile as sf
|
||||
import torch
|
||||
from fastapi import Depends, FastAPI, Header, HTTPException
|
||||
from fastapi.responses import Response
|
||||
from pydantic import BaseModel
|
||||
|
||||
logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s: %(message)s")
|
||||
log = logging.getLogger("qwen-tts")
|
||||
|
||||
MODEL = os.environ.get("QWEN_TTS_MODEL", "Qwen/Qwen3-TTS-12Hz-0.6B-CustomVoice")
|
||||
CLONE_MODEL = os.environ.get("QWEN_TTS_CLONE_MODEL", "Qwen/Qwen3-TTS-12Hz-0.6B-Base")
|
||||
VOICES_DIR = Path(os.environ.get("QWEN_TTS_VOICES_DIR", Path(__file__).parent / "voices"))
|
||||
DTYPE = getattr(torch, os.environ.get("QWEN_TTS_DTYPE", "float32"))
|
||||
THREADS = int(os.environ.get("QWEN_TTS_THREADS", "0")) or os.cpu_count() or 1
|
||||
API_KEY = os.environ.get("API_KEY", "")
|
||||
|
||||
# Voix prédéfinies du modèle CustomVoice (toutes parlent français)
|
||||
PRESETS: dict[str, tuple[str, str]] = {
|
||||
"serena": ("Serena — douce et chaleureuse", "female"),
|
||||
"vivian": ("Vivian — vive et lumineuse", "female"),
|
||||
"sohee": ("Sohee — expressive", "female"),
|
||||
"ono_anna": ("Anna — espiègle", "female"),
|
||||
"aiden": ("Aiden — solaire", "male"),
|
||||
"ryan": ("Ryan — dynamique", "male"),
|
||||
"uncle_fu": ("Fu — grave et posée", "male"),
|
||||
"dylan": ("Dylan — jeune et naturelle", "male"),
|
||||
"eric": ("Eric — légèrement voilée", "male"),
|
||||
}
|
||||
DEFAULT_PRESET = "serena"
|
||||
|
||||
torch.set_num_threads(THREADS)
|
||||
_lock = asyncio.Lock()
|
||||
_models: dict[str, object] = {}
|
||||
_clone_prompts: dict[str, object] = {}
|
||||
|
||||
|
||||
def _load(name: str):
|
||||
if name not in _models:
|
||||
from qwen_tts import Qwen3TTSModel
|
||||
|
||||
started = time.monotonic()
|
||||
_models[name] = Qwen3TTSModel.from_pretrained(name, device_map="cpu", dtype=DTYPE)
|
||||
log.info("Modèle %s chargé en %.1f s (%d threads)", name, time.monotonic() - started, THREADS)
|
||||
return _models[name]
|
||||
|
||||
|
||||
def clones() -> dict[str, dict]:
|
||||
"""Voix de référence déposées dans voices/ : nom.wav + nom.txt (+ nom.json)."""
|
||||
found = {}
|
||||
for wav in sorted(VOICES_DIR.glob("*.wav")):
|
||||
transcript = wav.with_suffix(".txt")
|
||||
if not transcript.exists():
|
||||
log.warning("Voix %s ignorée : transcription %s absente", wav.name, transcript.name)
|
||||
continue
|
||||
meta_file = wav.with_suffix(".json")
|
||||
meta = json.loads(meta_file.read_text()) if meta_file.exists() else {}
|
||||
found[wav.stem] = {
|
||||
"id": f"clone:{wav.stem}",
|
||||
"label": meta.get("label", wav.stem.replace("_", " ").title()),
|
||||
"gender": meta.get("gender", "female"),
|
||||
"wav": wav,
|
||||
"text": transcript.read_text().strip(),
|
||||
}
|
||||
return found
|
||||
|
||||
|
||||
def _generate(text: str, voice: str, instruct: str, language: str) -> tuple[bytes, str]:
|
||||
if voice.startswith("clone:") and CLONE_MODEL:
|
||||
name = voice.removeprefix("clone:")
|
||||
clone = clones().get(name)
|
||||
if clone is None:
|
||||
raise HTTPException(status_code=404, detail=f"Voix clonée inconnue : {name}")
|
||||
model = _load(CLONE_MODEL)
|
||||
if name not in _clone_prompts:
|
||||
_clone_prompts[name] = model.create_voice_clone_prompt(
|
||||
ref_audio=str(clone["wav"]), ref_text=clone["text"], x_vector_only_mode=False
|
||||
)
|
||||
wavs, rate = model.generate_voice_clone(
|
||||
text=text, language=language, voice_clone_prompt=_clone_prompts[name]
|
||||
)
|
||||
used = clone["id"]
|
||||
else:
|
||||
speaker = voice.lower() if voice.lower() in PRESETS else DEFAULT_PRESET
|
||||
kwargs = {"text": text, "language": language, "speaker": speaker}
|
||||
if instruct:
|
||||
kwargs["instruct"] = instruct
|
||||
wavs, rate = _load(MODEL).generate_custom_voice(**kwargs)
|
||||
used = speaker
|
||||
|
||||
buffer = io.BytesIO()
|
||||
sf.write(buffer, wavs[0], rate, format="WAV", subtype="PCM_16")
|
||||
return buffer.getvalue(), used
|
||||
|
||||
|
||||
@contextlib.asynccontextmanager
|
||||
async def lifespan(_app: FastAPI):
|
||||
# Chargé au démarrage : la première voix n'attend pas le chargement
|
||||
await asyncio.to_thread(_load, MODEL)
|
||||
log.info("Service prêt, %d voix clonée(s)", len(clones()))
|
||||
yield
|
||||
|
||||
|
||||
app = FastAPI(title="SocialFlow Qwen TTS", lifespan=lifespan)
|
||||
|
||||
|
||||
def require_key(x_api_key: str | None = Header(None)) -> None:
|
||||
if API_KEY and x_api_key != API_KEY:
|
||||
raise HTTPException(status_code=401, detail="Invalid API Key")
|
||||
|
||||
|
||||
class SynthesisRequest(BaseModel):
|
||||
text: str
|
||||
voice: str | None = None
|
||||
instruct: str | None = None
|
||||
language: str = "French"
|
||||
|
||||
|
||||
@app.get("/health", dependencies=[Depends(require_key)])
|
||||
async def health():
|
||||
return {"status": "ok", "model": MODEL, "busy": _lock.locked()}
|
||||
|
||||
|
||||
@app.get("/voices", dependencies=[Depends(require_key)])
|
||||
async def voices():
|
||||
presets = [{"id": k, "label": v[0], "gender": v[1]} for k, v in PRESETS.items()]
|
||||
cloned = [{k: c[k] for k in ("id", "label", "gender")} for c in clones().values()] if CLONE_MODEL else []
|
||||
return {"voices": [*cloned, *presets]}
|
||||
|
||||
|
||||
@app.post("/synthesize", dependencies=[Depends(require_key)])
|
||||
async def synthesize(request: SynthesisRequest):
|
||||
text = request.text.strip()
|
||||
if not text:
|
||||
raise HTTPException(status_code=400, detail="Texte vide")
|
||||
async with _lock:
|
||||
started = time.monotonic()
|
||||
audio, used = await asyncio.to_thread(
|
||||
_generate, text, request.voice or DEFAULT_PRESET, (request.instruct or "").strip(), request.language
|
||||
)
|
||||
log.info("Voix %s : %d car. en %.1f s", used, len(text), time.monotonic() - started)
|
||||
return Response(audio, media_type="audio/wav", headers={"X-Voice": used})
|
||||
@@ -0,0 +1,6 @@
|
||||
fastapi==0.115.12
|
||||
uvicorn[standard]==0.34.2
|
||||
pydantic==2.11.4
|
||||
qwen-tts==0.1.1
|
||||
soundfile==0.13.1
|
||||
numpy>=1.26
|
||||
@@ -0,0 +1,17 @@
|
||||
# Voix clonées (Qwen3-TTS)
|
||||
|
||||
Pour une voix française native, déposez ici un court extrait (5 à 15 s,
|
||||
une seule personne, sans musique) et sa transcription exacte :
|
||||
|
||||
```
|
||||
voices/
|
||||
camille.wav # extrait de référence
|
||||
camille.txt # texte prononcé dans l'extrait
|
||||
camille.json # optionnel : {"label": "Camille — chaleureuse", "gender": "female"}
|
||||
```
|
||||
|
||||
La voix apparaît alors dans le choix des voix Qwen sous l'identifiant
|
||||
`clone:camille`. Le dossier peut aussi être monté en volume
|
||||
(`QWEN_TTS_VOICES_DIR`) pour ajouter des voix sans reconstruire l'image.
|
||||
|
||||
Utilisez uniquement des voix dont vous avez les droits.
|
||||
@@ -271,6 +271,20 @@ reelsRouter.post('/reels/generate-text', async (req: Request, res: Response) =>
|
||||
}
|
||||
});
|
||||
|
||||
/**
|
||||
* Voix disponibles (dont les voix clonées Qwen) et disponibilité du moteur local
|
||||
* GET /api/reels/voices
|
||||
*/
|
||||
reelsRouter.get('/reels/voices', async (_req: Request, res: Response) => {
|
||||
try {
|
||||
const catalog = await ffmpegService.listVoices();
|
||||
res.json({ qwenAvailable: catalog.qwen_available, qwen: catalog.qwen });
|
||||
} catch (error) {
|
||||
console.error('❌ Error listing voices:', error);
|
||||
res.json({ qwenAvailable: false, qwen: [] });
|
||||
}
|
||||
});
|
||||
|
||||
/**
|
||||
* Prévisualiser la voix TTS
|
||||
* POST /api/reels/tts-preview
|
||||
|
||||
@@ -11,12 +11,13 @@ import { Readable } from 'stream';
|
||||
import { pipeline } from 'stream/promises';
|
||||
import type { ReadableStream as WebReadableStream } from 'stream/web';
|
||||
import type { SrtCue } from '@shared/srt';
|
||||
import type { TtsEngine, TtsStyle } from '@shared/voices';
|
||||
import type { TtsEngine, TtsStyle, VoiceOption } from '@shared/voices';
|
||||
|
||||
/** Un rendu long (stabilisation + encodage) peut dépasser plusieurs minutes. */
|
||||
const PROCESS_TIMEOUT_MS = 20 * 60_000;
|
||||
const DOWNLOAD_TIMEOUT_MS = 5 * 60_000;
|
||||
const TTS_TIMEOUT_MS = 3 * 60_000;
|
||||
// Qwen (local, CPU) met plusieurs fois la durée de la voix à la générer
|
||||
const TTS_TIMEOUT_MS = 15 * 60_000;
|
||||
const HEALTH_TIMEOUT_MS = 5_000;
|
||||
|
||||
export interface ReelRenderOptions {
|
||||
@@ -70,6 +71,13 @@ export interface TimedWord {
|
||||
end: number;
|
||||
}
|
||||
|
||||
export interface VoiceCatalog {
|
||||
qwen: VoiceOption[];
|
||||
qwen_available: boolean;
|
||||
gemini: VoiceOption[];
|
||||
edge: VoiceOption[];
|
||||
}
|
||||
|
||||
export interface VoicePreview {
|
||||
audio: Buffer;
|
||||
duration: number;
|
||||
@@ -285,6 +293,12 @@ export class FFmpegService {
|
||||
}
|
||||
}
|
||||
|
||||
/** Voix proposées par moteur ; `qwen_available` indique si le service local répond. */
|
||||
async listVoices(): Promise<VoiceCatalog> {
|
||||
const response = await this.call('/voices', { method: 'GET', timeoutMs: 15_000 });
|
||||
return await response.json() as VoiceCatalog;
|
||||
}
|
||||
|
||||
/** Génère la voix seule (aperçu), avec le minutage de chaque mot. */
|
||||
async previewVoice(
|
||||
text: string,
|
||||
|
||||
@@ -39,9 +39,12 @@ export async function resolveLogoPath(): Promise<string | undefined> {
|
||||
return undefined;
|
||||
}
|
||||
|
||||
/** Clé Gemini : configuration de l'application, sinon variable d'environnement. */
|
||||
/**
|
||||
* Clé Gemini : configuration de l'application, sinon variable d'environnement.
|
||||
* Aussi transmise avec Qwen, qui se replie sur Gemini si le service local échoue.
|
||||
*/
|
||||
export async function resolveGeminiApiKey(ttsEngine?: string): Promise<string | undefined> {
|
||||
if (ttsEngine !== "gemini") return undefined;
|
||||
if (ttsEngine !== "gemini" && ttsEngine !== "qwen") return undefined;
|
||||
const appConfig = await storage.getAppConfig();
|
||||
return appConfig?.geminiApiKey ?? process.env.GEMINI_API_KEY ?? undefined;
|
||||
}
|
||||
|
||||
+18
-3
@@ -3,7 +3,7 @@
|
||||
* Même catalogue que le service Python (ffmpeg-service/app/voices.py).
|
||||
*/
|
||||
|
||||
export const TTS_ENGINES = ["gemini", "edge"] as const;
|
||||
export const TTS_ENGINES = ["qwen", "gemini", "edge"] as const;
|
||||
export type TtsEngine = (typeof TTS_ENGINES)[number];
|
||||
|
||||
export const TTS_STYLES = ["neutral", "dynamic", "warm", "calm", "promo"] as const;
|
||||
@@ -16,7 +16,7 @@ export interface VoiceOption {
|
||||
}
|
||||
|
||||
export const TTS_STYLE_OPTIONS: { id: TtsStyle; label: string; description: string }[] = [
|
||||
{ id: "dynamic", label: "Dynamique", description: "Enthousiaste, rythme entraînant" },
|
||||
{ id: "dynamic", label: "Dynamique", description: "Enthousiaste, débit naturel" },
|
||||
{ id: "warm", label: "Chaleureux", description: "Souriant et proche" },
|
||||
{ id: "promo", label: "Promo", description: "Annonce énergique des offres" },
|
||||
{ id: "calm", label: "Calme", description: "Posé et rassurant" },
|
||||
@@ -64,13 +64,28 @@ export const EDGE_VOICES: VoiceOption[] = [
|
||||
{ id: "fr-FR-HenriNeural", label: "Henri", gender: "male" },
|
||||
];
|
||||
|
||||
/** Voix prédéfinies de Qwen3-TTS (local). Les voix clonées sont listées par le service. */
|
||||
export const QWEN_VOICES: VoiceOption[] = [
|
||||
{ id: "serena", label: "Serena — douce et chaleureuse", gender: "female" },
|
||||
{ id: "vivian", label: "Vivian — vive et lumineuse", gender: "female" },
|
||||
{ id: "sohee", label: "Sohee — expressive", gender: "female" },
|
||||
{ id: "ono_anna", label: "Anna — espiègle", gender: "female" },
|
||||
{ id: "aiden", label: "Aiden — solaire", gender: "male" },
|
||||
{ id: "ryan", label: "Ryan — dynamique", gender: "male" },
|
||||
{ id: "uncle_fu", label: "Fu — grave et posée", gender: "male" },
|
||||
{ id: "dylan", label: "Dylan — jeune et naturelle", gender: "male" },
|
||||
{ id: "eric", label: "Eric — légèrement voilée", gender: "male" },
|
||||
];
|
||||
|
||||
export const DEFAULT_VOICE: Record<TtsEngine, string> = {
|
||||
qwen: "serena",
|
||||
gemini: "Kore",
|
||||
edge: "fr-FR-VivienneMultilingualNeural",
|
||||
};
|
||||
|
||||
export const DEFAULT_TTS_STYLE: TtsStyle = "dynamic";
|
||||
export const DEFAULT_TTS_STYLE: TtsStyle = "neutral";
|
||||
|
||||
export function voicesFor(engine: TtsEngine): VoiceOption[] {
|
||||
if (engine === "qwen") return QWEN_VOICES;
|
||||
return engine === "gemini" ? GEMINI_VOICES : EDGE_VOICES;
|
||||
}
|
||||
Reference in new issue
Block a user