mirror of
https://github.com/R0m1k3/Loki.git
synced 2026-10-11 17:26:57 +02:00
Dictée : whisper-server supervisé, modèle chargé une seule fois
Chaque dictée relançait whisper-cli, qui relisait le modèle depuis le disque avant de transcrire. Ce chargement dominait le temps de réponse — 3,2 s de calcul pour 3,4 s d'audio — et aucun réglage ne pouvait le rattraper. whisper-server garde le modèle en mémoire entre deux phrases. Loki le supervise : démarrage au premier clic sur le micro, extinction après dix minutes sans dictée. Une dictée n'est pas un service permanent, et garder un modèle chargé toute la journée priverait le moteur de chat de sa VRAM. - Réglages serveur (modèle, langue, matériel, réactivité) dans bkState, avec leurs routes. La langue par défaut passe de « auto » à « fr » : sur quelques secondes d'audio la détection se trompe, et une langue mal détectée produit du charabia — des suites de caractères géorgiens ont été observées. - Catalogue de quatre modèles, de small à large-v3 ; large-v3-turbo par défaut. Téléchargement par identifiant, dans un .part renommé à la fin : un transfert interrompu ne laisse plus un .bin tronqué qu'on croirait bon. - Le GPU se choisit par CUDA_VISIBLE_DEVICES, posé en REMPLAÇANT toute valeur héritée. Dupliquée, la variable laisse le gagnant dépendre de la libc. - Les marqueurs de whisper ([BLANK_AUDIO], (silence)) ne sont plus collés dans le champ de saisie comme s'ils étaient du texte dicté. L'image construit DEUX binaires whisper-server, CPU et CUDA. LLAMACPP_IMAGE accepte la variante CPU de l'image amont ; sur cette base les .so CUDA sont absentes et un binaire lié à CUDA n'a même pas de quoi démarrer, donc aucun repli n'est possible depuis le programme. Le garde-fou du Dockerfile vérifie maintenant les deux étapes : les drapeaux de portabilité de la PR #18 doivent survivre à l'arrivée de CUDA. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01W3ewMAsXhkw9RY11kb9Dc9
This commit is contained in:
1 parent
3624194b39
commit
3cd4facb0e
11 files changed
+1222
-146
No files matched your search
+48
-22
@@ -31,29 +31,52 @@ RUN CGO_ENABLED=0 go build -trimpath -ldflags "-s -w" -o /out/loki ./cmd/loki
|
||||
# limité à cette étape de build.
|
||||
RUN GOBIN=/out GOTOOLCHAIN=auto go install golang.org/x/tools/gopls@latest
|
||||
|
||||
# ── Étape 1 bis : whisper-cli (dictée vocale) ───────────────────────────
|
||||
# whisper.cpp compilé CPU seul, en statique : un binaire unique, aucune .so à
|
||||
# trimballer, et pas de compilation CUDA (des minutes de build pour un gain
|
||||
# nul sur des dictées de quelques secondes). Ubuntu 22.04 : glibc plus
|
||||
# ancienne que l'image runtime, donc compatible quoi qu'elle embarque.
|
||||
# ── Étape 1 bis : whisper-server (dictée vocale) ────────────────────────
|
||||
# DEUX binaires sont construits, pas un. LLAMACPP_IMAGE accepte la variante
|
||||
# CPU de l'image amont (voir ligne 11) : sur cette base, les bibliothèques
|
||||
# CUDA sont absentes et un binaire lié à CUDA ne démarre pas du tout —
|
||||
# l'éditeur de liens échoue avant la première instruction, donc aucun repli
|
||||
# n'est possible depuis l'intérieur du programme. Loki choisit à l'exécution
|
||||
# (whisperServerBin, dictate_server.go).
|
||||
#
|
||||
# ⚠️ GGML_NATIVE=OFF est OBLIGATOIRE. Par défaut ggml compile en -march=native,
|
||||
# c'est-à-dire pour le processeur DU RUNNER DE BUILD — un Xeon récent chez
|
||||
# GitHub, avec AVX-512 et AMX. Le binaire partait alors sur une machine qui
|
||||
# n'a pas ces instructions et mourait d'un SIGILL en pleine transcription :
|
||||
# « AMX is not ready to be used! », puis plus rien, l'interface affichant un
|
||||
# échec sans raison. OFF retombe sur la ligne de base AVX2/FMA/F16C de ggml,
|
||||
# présente sur tout x86-64 depuis 2013.
|
||||
FROM ubuntu:22.04 AS whisperbuild
|
||||
# whisper-server et non whisper-cli : le modèle reste chargé entre deux
|
||||
# dictées. L'ancien chemin le relisait depuis le disque à chaque phrase, ce
|
||||
# qui dominait le temps de réponse.
|
||||
#
|
||||
# ⚠️ GGML_NATIVE=OFF est OBLIGATOIRE sur LES DEUX cibles. Par défaut ggml
|
||||
# compile en -march=native, c'est-à-dire pour le processeur DU RUNNER DE
|
||||
# BUILD — un Xeon récent chez GitHub, avec AVX-512 et AMX. Le binaire partait
|
||||
# alors sur une machine qui n'a pas ces instructions et mourait d'un SIGILL en
|
||||
# pleine transcription : « AMX is not ready to be used! », puis plus rien,
|
||||
# l'interface affichant un échec sans raison. OFF retombe sur la ligne de base
|
||||
# AVX2/FMA/F16C de ggml, présente sur tout x86-64 depuis 2013.
|
||||
ARG WHISPER_CMAKE_FLAGS="-DCMAKE_BUILD_TYPE=Release -DBUILD_SHARED_LIBS=OFF \
|
||||
-DWHISPER_BUILD_TESTS=OFF -DWHISPER_BUILD_EXAMPLES=ON \
|
||||
-DGGML_NATIVE=OFF -DGGML_AVX512=OFF -DGGML_AMX_TILE=OFF \
|
||||
-DGGML_AMX_INT8=OFF -DGGML_AMX_BF16=OFF"
|
||||
|
||||
# Ubuntu 22.04 : glibc plus ancienne que l'image runtime, donc compatible quoi
|
||||
# qu'elle embarque.
|
||||
FROM ubuntu:22.04 AS whisperbuild-cpu
|
||||
ARG WHISPER_CMAKE_FLAGS
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
build-essential cmake git ca-certificates \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
RUN git clone --depth 1 https://github.com/ggml-org/whisper.cpp /w \
|
||||
&& cmake -S /w -B /w/build -DCMAKE_BUILD_TYPE=Release \
|
||||
-DBUILD_SHARED_LIBS=OFF -DWHISPER_BUILD_TESTS=OFF \
|
||||
-DGGML_NATIVE=OFF -DGGML_AVX512=OFF -DGGML_AMX_TILE=OFF \
|
||||
-DGGML_AMX_INT8=OFF -DGGML_AMX_BF16=OFF \
|
||||
&& cmake --build /w/build -j --target whisper-cli
|
||||
&& cmake -S /w -B /w/build ${WHISPER_CMAKE_FLAGS} \
|
||||
&& cmake --build /w/build -j --target whisper-server
|
||||
|
||||
# Version CUDA. La MAJEURE de CUDA doit correspondre à celle de l'image
|
||||
# runtime : le binaire est lié dynamiquement à libcudart, fournie par
|
||||
# LLAMACPP_IMAGE et non par cette étape.
|
||||
FROM nvidia/cuda:12.4.1-devel-ubuntu22.04 AS whisperbuild-cuda
|
||||
ARG WHISPER_CMAKE_FLAGS
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
build-essential cmake git ca-certificates \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
RUN git clone --depth 1 https://github.com/ggml-org/whisper.cpp /w \
|
||||
&& cmake -S /w -B /w/build ${WHISPER_CMAKE_FLAGS} -DGGML_CUDA=ON \
|
||||
&& cmake --build /w/build -j --target whisper-server
|
||||
|
||||
# ── Étape 2 : runtime sur l'image serveur CUDA officielle ───────────────
|
||||
FROM ${LLAMACPP_IMAGE} AS runtime
|
||||
@@ -108,9 +131,11 @@ RUN if [ "$PLAYWRIGHT" = "1" ]; then \
|
||||
|
||||
COPY --from=gobuild /out/loki /usr/local/bin/loki
|
||||
COPY --from=gobuild /out/gopls /usr/local/bin/gopls
|
||||
# Dictée vocale (POST /api/transcribe). Le modèle (~190 Mo) n'est PAS dans
|
||||
# l'image : téléchargé au premier usage dans /data/whisper/.
|
||||
COPY --from=whisperbuild /w/build/bin/whisper-cli /usr/local/bin/whisper-cli
|
||||
# Dictée vocale (POST /api/transcribe). Les modèles (190 Mo à 1,1 Go) ne sont
|
||||
# PAS dans l'image : téléchargés à la demande dans /data/whisper/.
|
||||
# Deux binaires : voir l'étape whisperbuild-cpu pour la raison.
|
||||
COPY --from=whisperbuild-cpu /w/build/bin/whisper-server /usr/local/bin/whisper-server-cpu
|
||||
COPY --from=whisperbuild-cuda /w/build/bin/whisper-server /usr/local/bin/whisper-server-cuda
|
||||
COPY docker-entrypoint.sh /usr/local/bin/docker-entrypoint.sh
|
||||
RUN chmod +x /usr/local/bin/docker-entrypoint.sh && mkdir -p /data /models
|
||||
|
||||
@@ -119,7 +144,8 @@ ENV LOKI_CONTAINER=1 \
|
||||
LOKI_HOME=/data \
|
||||
LOKI_MODEL_DIRS=/models \
|
||||
LOKI_ENGINE_BIN=/app/llama-server \
|
||||
LOKI_WHISPER_BIN=/usr/local/bin/whisper-cli \
|
||||
LOKI_WHISPER_SERVER_CPU=/usr/local/bin/whisper-server-cpu \
|
||||
LOKI_WHISPER_SERVER_CUDA=/usr/local/bin/whisper-server-cuda \
|
||||
LD_LIBRARY_PATH=/app \
|
||||
NVIDIA_VISIBLE_DEVICES=all \
|
||||
NVIDIA_DRIVER_CAPABILITIES=compute,utility
|
||||
|
||||
@@ -0,0 +1,90 @@
|
||||
// Réglages de la dictée vocale — modèle, langue, matériel, réactivité.
|
||||
//
|
||||
// Ils vivent dans le magasin clé/valeur (bkState), comme le jeton Hugging
|
||||
// Face : ce sont des réglages de SERVEUR, pas de navigateur. Le modèle chargé
|
||||
// et le GPU occupé sont des propriétés de la machine, identiques pour tous les
|
||||
// onglets ouverts.
|
||||
package loki
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strconv"
|
||||
"time"
|
||||
)
|
||||
|
||||
const dictateCfgKey = "dictate_cfg"
|
||||
|
||||
// DictateCfg : Device vaut "cpu" ou un indice de GPU en décimal ("0", "1"),
|
||||
// dans l'ordre de nvidia-smi — le même que celui affiché par /api/vram.
|
||||
type DictateCfg struct {
|
||||
Model string `json:"model"`
|
||||
Lang string `json:"lang"`
|
||||
Device string `json:"device"`
|
||||
Reactivity string `json:"reactivity"`
|
||||
}
|
||||
|
||||
// withDefauts comble les champs vides. Appliqué à la LECTURE plutôt qu'à
|
||||
// l'écriture : une version future qui ajoute un champ trouvera les réglages
|
||||
// déjà enregistrés sans champ correspondant, et doit pouvoir le combler.
|
||||
func (c DictateCfg) withDefauts() DictateCfg {
|
||||
if c.Model == "" {
|
||||
// large-v3-turbo : meilleur rapport qualité/vitesse en français à ce
|
||||
// jour. Le défaut penche vers la qualité.
|
||||
c.Model = "large-v3-turbo-q5_0"
|
||||
}
|
||||
if c.Lang == "" {
|
||||
// Pas "auto" : sur des tranches de quelques secondes la détection se
|
||||
// trompe, et une langue mal détectée produit du charabia — des suites
|
||||
// de caractères géorgiens ont été observées en production.
|
||||
c.Lang = "fr"
|
||||
}
|
||||
if c.Device == "" {
|
||||
c.Device = "cpu"
|
||||
}
|
||||
if c.Reactivity == "" {
|
||||
c.Reactivity = "moyen"
|
||||
}
|
||||
return c
|
||||
}
|
||||
|
||||
func dictateCfgLoad() DictateCfg {
|
||||
var c DictateCfg
|
||||
getJSON(bkState, dictateCfgKey, &c)
|
||||
return c.withDefauts()
|
||||
}
|
||||
|
||||
func dictateCfgSave(c DictateCfg) error {
|
||||
c = c.withDefauts()
|
||||
// Valider ICI plutôt qu'au démarrage du serveur : un modèle hors catalogue
|
||||
// accepté en silence ne se manifesterait qu'au premier clic sur le micro,
|
||||
// loin du geste qui l'a causé.
|
||||
if _, ok := whisperCatalogue[c.Model]; !ok {
|
||||
return fmt.Errorf("modèle de dictée inconnu : %s", c.Model)
|
||||
}
|
||||
return putJSON(bkState, dictateCfgKey, c)
|
||||
}
|
||||
|
||||
// cudaVisibleDevices traduit le réglage en valeur de CUDA_VISIBLE_DEVICES.
|
||||
// Vide = aucun GPU visible = whisper tourne sur le CPU. Toute valeur qui n'est
|
||||
// pas un indice positif retombe sur le CPU : mieux vaut une dictée lente qu'un
|
||||
// GPU choisi au hasard sur une machine dont on ne sait rien.
|
||||
func (c DictateCfg) cudaVisibleDevices() string {
|
||||
n, err := strconv.Atoi(c.Device)
|
||||
if err != nil || n < 0 {
|
||||
return ""
|
||||
}
|
||||
return strconv.Itoa(n)
|
||||
}
|
||||
|
||||
// chunkBounds : réserve minimale avant de couper, et coupure forcée quand
|
||||
// aucun silence n'arrive. Consommé par le découpeur du plan « temps réel ».
|
||||
func (c DictateCfg) chunkBounds() (time.Duration, time.Duration) {
|
||||
switch c.Reactivity {
|
||||
case "court":
|
||||
return 1000 * time.Millisecond, 5 * time.Second
|
||||
case "long":
|
||||
return 3 * time.Second, 15 * time.Second
|
||||
default:
|
||||
return 1500 * time.Millisecond, 8 * time.Second
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,87 @@
|
||||
package loki
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
func TestDictateCfgDefauts(t *testing.T) {
|
||||
c := DictateCfg{}.withDefauts()
|
||||
if c.Model != "large-v3-turbo-q5_0" {
|
||||
t.Errorf("modèle par défaut = %q, attendu large-v3-turbo-q5_0", c.Model)
|
||||
}
|
||||
if c.Lang != "fr" {
|
||||
t.Errorf("langue par défaut = %q, attendu fr — la détection auto se trompe sur les tranches courtes", c.Lang)
|
||||
}
|
||||
if c.Device != "cpu" {
|
||||
t.Errorf("matériel par défaut = %q, attendu cpu (aucun GPU supposé)", c.Device)
|
||||
}
|
||||
if c.Reactivity != "moyen" {
|
||||
t.Errorf("réactivité par défaut = %q, attendu moyen", c.Reactivity)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCudaVisibleDevices(t *testing.T) {
|
||||
cas := []struct{ device, want string }{
|
||||
{"cpu", ""},
|
||||
{"0", "0"},
|
||||
{"1", "1"},
|
||||
{"", ""},
|
||||
{"bidon", ""},
|
||||
{"-1", ""},
|
||||
}
|
||||
for _, c := range cas {
|
||||
got := DictateCfg{Device: c.device}.cudaVisibleDevices()
|
||||
if got != c.want {
|
||||
t.Errorf("cudaVisibleDevices(%q) = %q, attendu %q", c.device, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestChunkBounds(t *testing.T) {
|
||||
cas := []struct {
|
||||
react string
|
||||
min, max time.Duration
|
||||
}{
|
||||
{"court", 1000 * time.Millisecond, 5 * time.Second},
|
||||
{"moyen", 1500 * time.Millisecond, 8 * time.Second},
|
||||
{"long", 3 * time.Second, 15 * time.Second},
|
||||
{"inconnu", 1500 * time.Millisecond, 8 * time.Second},
|
||||
}
|
||||
for _, c := range cas {
|
||||
min, max := DictateCfg{Reactivity: c.react}.chunkBounds()
|
||||
if min != c.min || max != c.max {
|
||||
t.Errorf("chunkBounds(%q) = %v/%v, attendu %v/%v", c.react, min, max, c.min, c.max)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestDictateCfgAllerRetour(t *testing.T) {
|
||||
testHome(t)
|
||||
in := DictateCfg{Model: "medium-q5_0", Lang: "en", Device: "1", Reactivity: "long"}
|
||||
if err := dictateCfgSave(in); err != nil {
|
||||
t.Fatalf("enregistrement : %v", err)
|
||||
}
|
||||
if out := dictateCfgLoad(); out != in {
|
||||
t.Errorf("relu %+v, attendu %+v", out, in)
|
||||
}
|
||||
}
|
||||
|
||||
// Un modèle hors catalogue accepté en silence ne se manifesterait qu'au premier
|
||||
// clic sur le micro, loin du geste qui l'a causé.
|
||||
func TestDictateCfgSaveRefuseModeleInconnu(t *testing.T) {
|
||||
testHome(t)
|
||||
if err := dictateCfgSave(DictateCfg{Model: "modele-inexistant", Lang: "fr", Device: "cpu", Reactivity: "moyen"}); err == nil {
|
||||
t.Error("un modèle hors catalogue doit être refusé")
|
||||
}
|
||||
}
|
||||
|
||||
// Sans réglage enregistré, la lecture doit rendre les défauts — et non le zéro
|
||||
// de la structure, qui donnerait un modèle vide à whisper-server.
|
||||
func TestDictateCfgLoadSansRien(t *testing.T) {
|
||||
testHome(t)
|
||||
c := dictateCfgLoad()
|
||||
if c.Model == "" || c.Lang == "" || c.Device == "" || c.Reactivity == "" {
|
||||
t.Errorf("lecture à vide = %+v, aucun champ ne doit rester vide", c)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,74 @@
|
||||
// Catalogue des modèles de dictée. Aucun n'est embarqué dans l'image : ils
|
||||
// pèsent de 190 Mo à 1,1 Go et vivent dans <LOKI_HOME>/whisper/, donc dans le
|
||||
// volume /data — ils survivent aux recréations du conteneur.
|
||||
package loki
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
)
|
||||
|
||||
// Les modèles ggml officiels de whisper.cpp.
|
||||
const whisperModelBase = "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/"
|
||||
|
||||
type whisperModel struct {
|
||||
Nom string // libellé affiché
|
||||
Fichier string // nom du .bin, identique en local et chez Hugging Face
|
||||
Octets int64 // taille du téléchargement, pour prévenir avant de le lancer
|
||||
VRAMMo int // ordre de grandeur occupé sur le GPU, pour choisir en connaissance de cause
|
||||
}
|
||||
|
||||
// Les tailles sont celles publiées par le dépôt amont. Elles servent à
|
||||
// informer avant un téléchargement, pas à vérifier le fichier : une taille qui
|
||||
// dérive d'une release à l'autre ne doit pas casser la dictée.
|
||||
var whisperCatalogue = map[string]whisperModel{
|
||||
"small-q5_1": {Nom: "small — rapide, correct", Fichier: "ggml-small-q5_1.bin", Octets: 190_000_000, VRAMMo: 600},
|
||||
"medium-q5_0": {Nom: "medium — bon compromis", Fichier: "ggml-medium-q5_0.bin", Octets: 539_000_000, VRAMMo: 1400},
|
||||
"large-v3-turbo-q5_0": {Nom: "large-v3-turbo — recommandé", Fichier: "ggml-large-v3-turbo-q5_0.bin", Octets: 574_000_000, VRAMMo: 1600},
|
||||
"large-v3-q5_0": {Nom: "large-v3 — le plus précis, le plus lent", Fichier: "ggml-large-v3-q5_0.bin", Octets: 1_080_000_000, VRAMMo: 3600},
|
||||
}
|
||||
|
||||
// whisperModelPathFor : chemin local. Vide si l'identifiant est inconnu —
|
||||
// l'appelant doit traiter ce cas plutôt que de bâtir un chemin sur une chaîne
|
||||
// arbitraire venue d'une requête HTTP.
|
||||
func whisperModelPathFor(id string) string {
|
||||
m, ok := whisperCatalogue[id]
|
||||
if !ok {
|
||||
return ""
|
||||
}
|
||||
return filepath.Join(LokiHome(), "whisper", m.Fichier)
|
||||
}
|
||||
|
||||
func whisperModelURLFor(id string) string {
|
||||
m, ok := whisperCatalogue[id]
|
||||
if !ok {
|
||||
return ""
|
||||
}
|
||||
return whisperModelBase + m.Fichier
|
||||
}
|
||||
|
||||
func whisperModelPresent(id string) bool {
|
||||
p := whisperModelPathFor(id)
|
||||
if p == "" {
|
||||
return false
|
||||
}
|
||||
st, err := os.Stat(p)
|
||||
return err == nil && st.Size() > 0
|
||||
}
|
||||
|
||||
// whisperCatalogueTrie : le catalogue pour l'UI, du plus léger au plus lourd —
|
||||
// l'ordre dans lequel se pose la question du compromis.
|
||||
func whisperCatalogueTrie() []map[string]any {
|
||||
out := make([]map[string]any, 0, len(whisperCatalogue))
|
||||
for id, m := range whisperCatalogue {
|
||||
out = append(out, map[string]any{
|
||||
"id": id, "nom": m.Nom, "octets": m.Octets,
|
||||
"vram_mo": m.VRAMMo, "present": whisperModelPresent(id),
|
||||
})
|
||||
}
|
||||
sort.Slice(out, func(i, j int) bool {
|
||||
return out[i]["octets"].(int64) < out[j]["octets"].(int64)
|
||||
})
|
||||
return out
|
||||
}
|
||||
@@ -0,0 +1,75 @@
|
||||
package loki
|
||||
|
||||
import (
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestCatalogueContientLeDefautEtLAncien(t *testing.T) {
|
||||
// Le défaut doit exister, sinon dictateCfgSave refuse sa propre valeur par
|
||||
// défaut et plus aucun réglage n'est enregistrable.
|
||||
if _, ok := whisperCatalogue["large-v3-turbo-q5_0"]; !ok {
|
||||
t.Error("le modèle par défaut large-v3-turbo-q5_0 est absent du catalogue")
|
||||
}
|
||||
// small-q5_1 est le modèle déjà téléchargé sur les installations
|
||||
// existantes : le retirer le rendrait insélectionnable alors que son
|
||||
// fichier est là.
|
||||
if _, ok := whisperCatalogue["small-q5_1"]; !ok {
|
||||
t.Error("small-q5_1 est absent : les installations existantes l'ont déjà sur disque")
|
||||
}
|
||||
}
|
||||
|
||||
func TestModelPathEtURL(t *testing.T) {
|
||||
testHome(t)
|
||||
p := whisperModelPathFor("small-q5_1")
|
||||
if filepath.Base(p) != "ggml-small-q5_1.bin" {
|
||||
t.Errorf("chemin = %q, le fichier doit s'appeler ggml-small-q5_1.bin", p)
|
||||
}
|
||||
// Le chemin DOIT rester celui d'avant, sinon les installations existantes
|
||||
// re-téléchargent 190 Mo pour rien.
|
||||
if filepath.Base(filepath.Dir(p)) != "whisper" {
|
||||
t.Errorf("chemin = %q, le dossier doit rester <LOKI_HOME>/whisper", p)
|
||||
}
|
||||
u := whisperModelURLFor("small-q5_1")
|
||||
if !strings.HasSuffix(u, "/ggml-small-q5_1.bin") {
|
||||
t.Errorf("URL = %q, doit finir par /ggml-small-q5_1.bin", u)
|
||||
}
|
||||
if !strings.HasPrefix(u, "https://") {
|
||||
t.Errorf("URL = %q, doit être en HTTPS", u)
|
||||
}
|
||||
}
|
||||
|
||||
func TestModelPathInconnuVide(t *testing.T) {
|
||||
testHome(t)
|
||||
if p := whisperModelPathFor("nawak"); p != "" {
|
||||
t.Errorf("un modèle hors catalogue doit donner un chemin vide, pas %q", p)
|
||||
}
|
||||
if u := whisperModelURLFor("nawak"); u != "" {
|
||||
t.Errorf("URL d'un modèle inconnu = %q, attendu vide", u)
|
||||
}
|
||||
if whisperModelPresent("nawak") {
|
||||
t.Error("un modèle inconnu ne peut pas être présent")
|
||||
}
|
||||
}
|
||||
|
||||
func TestCatalogueTrieParTaille(t *testing.T) {
|
||||
testHome(t)
|
||||
l := whisperCatalogueTrie()
|
||||
if len(l) < 4 {
|
||||
t.Fatalf("catalogue de %d entrées, au moins 4 attendues", len(l))
|
||||
}
|
||||
var prec int64
|
||||
for i, m := range l {
|
||||
o, _ := m["octets"].(int64)
|
||||
if i > 0 && o < prec {
|
||||
t.Errorf("entrée %d : catalogue non trié par taille croissante", i)
|
||||
}
|
||||
prec = o
|
||||
for _, clef := range []string{"id", "nom", "octets", "vram_mo", "present"} {
|
||||
if _, ok := m[clef]; !ok {
|
||||
t.Errorf("entrée %d : clé %q manquante — l'UI en a besoin", i, clef)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,105 @@
|
||||
package loki
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"net/http/httptest"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestConfigRouteAllerRetour(t *testing.T) {
|
||||
testHome(t)
|
||||
corps := strings.NewReader(`{"model":"small-q5_1","lang":"en","device":"1","reactivity":"court"}`)
|
||||
rec := httptest.NewRecorder()
|
||||
handleDictateConfig(rec, httptest.NewRequest("POST", "/api/dictate/config", corps))
|
||||
if rec.Code != 200 {
|
||||
t.Fatalf("POST = %d, corps %s", rec.Code, rec.Body.String())
|
||||
}
|
||||
rec = httptest.NewRecorder()
|
||||
handleDictateConfig(rec, httptest.NewRequest("GET", "/api/dictate/config", nil))
|
||||
var got DictateCfg
|
||||
if err := json.Unmarshal(rec.Body.Bytes(), &got); err != nil {
|
||||
t.Fatalf("réponse illisible : %v", err)
|
||||
}
|
||||
if got.Device != "1" || got.Lang != "en" || got.Model != "small-q5_1" || got.Reactivity != "court" {
|
||||
t.Errorf("relu %+v, attendu le réglage enregistré", got)
|
||||
}
|
||||
}
|
||||
|
||||
// Le refus doit venir de la ROUTE, pas seulement du magasin : sinon l'UI reçoit
|
||||
// un 200 et croit son réglage pris en compte.
|
||||
func TestConfigRouteRefuseModeleInconnu(t *testing.T) {
|
||||
testHome(t)
|
||||
corps := strings.NewReader(`{"model":"nawak","lang":"fr","device":"cpu","reactivity":"moyen"}`)
|
||||
rec := httptest.NewRecorder()
|
||||
handleDictateConfig(rec, httptest.NewRequest("POST", "/api/dictate/config", corps))
|
||||
if rec.Code == 200 {
|
||||
t.Errorf("modèle hors catalogue accepté (%d) : %s", rec.Code, rec.Body.String())
|
||||
}
|
||||
}
|
||||
|
||||
func TestConfigRouteCorpsIllisible(t *testing.T) {
|
||||
testHome(t)
|
||||
rec := httptest.NewRecorder()
|
||||
handleDictateConfig(rec, httptest.NewRequest("POST", "/api/dictate/config", strings.NewReader("pas du json")))
|
||||
if rec.Code != 400 {
|
||||
t.Errorf("code = %d, attendu 400", rec.Code)
|
||||
}
|
||||
}
|
||||
|
||||
func TestModelsRoute(t *testing.T) {
|
||||
testHome(t)
|
||||
rec := httptest.NewRecorder()
|
||||
handleDictateModels(rec, httptest.NewRequest("GET", "/api/dictate/models", nil))
|
||||
if rec.Code != 200 {
|
||||
t.Fatalf("GET = %d", rec.Code)
|
||||
}
|
||||
var l []map[string]any
|
||||
if err := json.Unmarshal(rec.Body.Bytes(), &l); err != nil {
|
||||
t.Fatalf("réponse illisible : %v", err)
|
||||
}
|
||||
if len(l) < 4 {
|
||||
t.Errorf("%d modèles renvoyés, au moins 4 attendus", len(l))
|
||||
}
|
||||
}
|
||||
|
||||
// Un identifiant inconnu ne doit pas lancer de téléchargement : whisperDownload
|
||||
// bâtirait une URL vide et le message d'erreur n'arriverait qu'en arrière-plan.
|
||||
func TestDownloadRouteRefuseInconnu(t *testing.T) {
|
||||
testHome(t)
|
||||
rec := httptest.NewRecorder()
|
||||
corps := strings.NewReader(`{"id":"nawak"}`)
|
||||
handleDictateDownload(rec, httptest.NewRequest("POST", "/api/dictate/models/download", corps))
|
||||
if rec.Code != 400 {
|
||||
t.Errorf("code = %d, attendu 400 pour un modèle inconnu", rec.Code)
|
||||
}
|
||||
}
|
||||
|
||||
func TestStateRoute(t *testing.T) {
|
||||
testHome(t)
|
||||
rec := httptest.NewRecorder()
|
||||
handleDictateState(rec, httptest.NewRequest("GET", "/api/dictate/state", nil))
|
||||
if rec.Code != 200 {
|
||||
t.Fatalf("GET = %d", rec.Code)
|
||||
}
|
||||
var e map[string]any
|
||||
if err := json.Unmarshal(rec.Body.Bytes(), &e); err != nil {
|
||||
t.Fatalf("réponse illisible : %v", err)
|
||||
}
|
||||
for _, clef := range []string{"actif", "modele", "present", "device", "dl"} {
|
||||
if _, ok := e[clef]; !ok {
|
||||
t.Errorf("clé %q manquante dans l'état", clef)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Un audio trop court est rejeté AVANT toute tentative de lancer whisper-server :
|
||||
// inutile de charger un modèle pour un clic parasite.
|
||||
func TestTranscribeAudioTropCourt(t *testing.T) {
|
||||
testHome(t)
|
||||
rec := httptest.NewRecorder()
|
||||
handleTranscribe(rec, httptest.NewRequest("POST", "/api/transcribe", strings.NewReader("court")))
|
||||
if rec.Code != 400 {
|
||||
t.Errorf("code = %d, attendu 400", rec.Code)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,328 @@
|
||||
// Supervision de whisper-server : le modèle est chargé UNE fois et reste en
|
||||
// mémoire, au lieu d'être relu depuis le disque à chaque phrase dictée.
|
||||
//
|
||||
// L'ancien chemin lançait whisper-cli par requête. Le chargement du modèle
|
||||
// dominait le temps de réponse (~3 s pour 3,4 s d'audio), et découper la
|
||||
// dictée en tranches par-dessus ce mécanisme aurait rechargé le modèle toutes
|
||||
// les quelques secondes.
|
||||
package loki
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"io"
|
||||
"mime/multipart"
|
||||
"net"
|
||||
"net/http"
|
||||
"os"
|
||||
"os/exec"
|
||||
"strconv"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
)
|
||||
|
||||
// whisperServerBin : deux binaires sont livrés, pas un.
|
||||
//
|
||||
// L'image runtime peut être bâtie sur la variante CPU de llama.cpp
|
||||
// (LLAMACPP_IMAGE=ghcr.io/ggml-org/llama.cpp:server). Sur cette base les
|
||||
// bibliothèques CUDA sont absentes, et un binaire lié à CUDA ne démarre pas du
|
||||
// tout : l'éditeur de liens échoue avant la première instruction, il n'y a
|
||||
// donc aucun repli possible depuis l'intérieur du programme. D'où deux
|
||||
// binaires et un choix fait ici.
|
||||
func whisperServerBin(gpu bool) string {
|
||||
env, defaut := "LOKI_WHISPER_SERVER_CPU", "/usr/local/bin/whisper-server-cpu"
|
||||
if gpu {
|
||||
env, defaut = "LOKI_WHISPER_SERVER_CUDA", "/usr/local/bin/whisper-server-cuda"
|
||||
}
|
||||
for _, p := range []string{os.Getenv(env), defaut} {
|
||||
if p == "" {
|
||||
continue
|
||||
}
|
||||
if _, err := os.Stat(p); err == nil {
|
||||
return p
|
||||
}
|
||||
}
|
||||
if gpu {
|
||||
// Pas de binaire CUDA : l'appelant retombe sur le CPU en le disant.
|
||||
return ""
|
||||
}
|
||||
if p, err := exec.LookPath("whisper-server"); err == nil {
|
||||
return p
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
func whisperServerArgs(c DictateCfg, port int) ([]string, error) {
|
||||
model := whisperModelPathFor(c.Model)
|
||||
if model == "" {
|
||||
return nil, fmt.Errorf("modèle de dictée inconnu : %s", c.Model)
|
||||
}
|
||||
return []string{
|
||||
"-m", model,
|
||||
"-l", c.Lang,
|
||||
// 127.0.0.1 et pas 0.0.0.0 : whisper-server n'a aucune
|
||||
// authentification. Sur un conteneur au réseau partagé, l'exposer
|
||||
// offrirait la transcription à tout le réseau.
|
||||
"--host", "127.0.0.1",
|
||||
"--port", strconv.Itoa(port),
|
||||
}, nil
|
||||
}
|
||||
|
||||
// whisperServerEnv pose CUDA_VISIBLE_DEVICES en REMPLAÇANT toute valeur déjà
|
||||
// présente. Une variable dupliquée laisse le gagnant dépendre de la libc, et
|
||||
// le choix de GPU deviendrait un coup de dé.
|
||||
func whisperServerEnv(c DictateCfg, base []string) []string {
|
||||
const clef = "CUDA_VISIBLE_DEVICES="
|
||||
out := make([]string, 0, len(base)+1)
|
||||
for _, v := range base {
|
||||
if !strings.HasPrefix(v, clef) {
|
||||
out = append(out, v)
|
||||
}
|
||||
}
|
||||
// Valeur vide en mode CPU : masque TOUS les GPU. Retirer la variable
|
||||
// ferait l'inverse — ggml les verrait tous.
|
||||
return append(out, clef+c.cudaVisibleDevices())
|
||||
}
|
||||
|
||||
// portLibre demande au système un port disponible puis le relâche. Coder un
|
||||
// port en dur casserait sur une machine qui l'occupe déjà ; la fenêtre entre
|
||||
// la fermeture et la reprise par whisper-server est négligeable devant le
|
||||
// risque d'un conflit permanent.
|
||||
func portLibre() (int, error) {
|
||||
l, err := net.Listen("tcp", "127.0.0.1:0")
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
defer func() { _ = l.Close() }()
|
||||
return l.Addr().(*net.TCPAddr).Port, nil
|
||||
}
|
||||
|
||||
// whisperMarqueur reconnaît les annotations que whisper produit à la place
|
||||
// d'un texte : [BLANK_AUDIO], (silence), [SOUND]… Un contenu entièrement
|
||||
// entouré de crochets ou de parenthèses n'est jamais de la parole dictée, et
|
||||
// le laisser passer le collerait tel quel dans le champ de saisie.
|
||||
func whisperMarqueur(s string) bool {
|
||||
s = strings.TrimSpace(s)
|
||||
if s == "" {
|
||||
return true
|
||||
}
|
||||
return (strings.HasPrefix(s, "[") && strings.HasSuffix(s, "]")) ||
|
||||
(strings.HasPrefix(s, "(") && strings.HasSuffix(s, ")"))
|
||||
}
|
||||
|
||||
// whisperInferSur prend l'URL de base en paramètre : c'est ce qui rend l'appel
|
||||
// testable sans lancer de processus.
|
||||
func whisperInferSur(ctx context.Context, base string, wav []byte) (string, error) {
|
||||
var corps bytes.Buffer
|
||||
mw := multipart.NewWriter(&corps)
|
||||
part, err := mw.CreateFormFile("file", "dictee.wav")
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
if _, err := part.Write(wav); err != nil {
|
||||
return "", err
|
||||
}
|
||||
if err := mw.WriteField("response_format", "json"); err != nil {
|
||||
return "", err
|
||||
}
|
||||
if err := mw.Close(); err != nil {
|
||||
return "", err
|
||||
}
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodPost, base+"/inference", &corps)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
req.Header.Set("Content-Type", mw.FormDataContentType())
|
||||
resp, err := http.DefaultClient.Do(req)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
defer func() { _ = resp.Body.Close() }()
|
||||
brut, _ := io.ReadAll(io.LimitReader(resp.Body, 1<<20))
|
||||
if resp.StatusCode != 200 {
|
||||
return "", fmt.Errorf("whisper-server a répondu %d : %s", resp.StatusCode, lastLine(string(brut)))
|
||||
}
|
||||
var j struct {
|
||||
Text string `json:"text"`
|
||||
}
|
||||
if err := json.Unmarshal(brut, &j); err != nil {
|
||||
return "", fmt.Errorf("réponse illisible de whisper-server : %w", err)
|
||||
}
|
||||
if whisperMarqueur(j.Text) {
|
||||
return "", nil
|
||||
}
|
||||
return strings.TrimSpace(j.Text), nil
|
||||
}
|
||||
|
||||
// ─── Supervision ─────────────────────────────────────────────────────────────
|
||||
|
||||
// whisperIdle : au-delà, le serveur s'éteint et rend sa VRAM. Une dictée n'est
|
||||
// pas un service permanent ; garder un modèle chargé toute la journée pour
|
||||
// quelques phrases prive le moteur de chat de mémoire GPU.
|
||||
const whisperIdle = 10 * time.Minute
|
||||
|
||||
// whisperReady : plafond d'attente au démarrage. Un modèle de 1 Go lu depuis un
|
||||
// disque lent puis poussé sur le GPU prend du temps ; échouer trop tôt
|
||||
// afficherait une panne là où il n'y a qu'une lenteur.
|
||||
const whisperReady = 120 * time.Second
|
||||
|
||||
var wsrvMu sync.Mutex
|
||||
var wsrv struct {
|
||||
cmd *exec.Cmd
|
||||
port int
|
||||
cfg DictateCfg
|
||||
log *bytes.Buffer
|
||||
timer *time.Timer
|
||||
note string // pourquoi on ne tourne pas sur le matériel demandé
|
||||
depuis time.Time
|
||||
}
|
||||
|
||||
// whisperEnsure garantit un serveur vivant pour la configuration COURANTE, et
|
||||
// renvoie son port. Si les réglages ont changé depuis le lancement, l'ancien
|
||||
// est arrêté : c'est ce qui rend un changement de modèle ou de GPU effectif
|
||||
// sans redémarrer Loki.
|
||||
func whisperEnsure() (int, error) {
|
||||
cfg := dictateCfgLoad()
|
||||
wsrvMu.Lock()
|
||||
defer wsrvMu.Unlock()
|
||||
if wsrv.cmd != nil && wsrv.cfg == cfg && wsrv.cmd.ProcessState == nil {
|
||||
whisperTouchLocked()
|
||||
return wsrv.port, nil
|
||||
}
|
||||
whisperStopLocked()
|
||||
return whisperStartLocked(cfg)
|
||||
}
|
||||
|
||||
func whisperStartLocked(cfg DictateCfg) (int, error) {
|
||||
if !whisperModelPresent(cfg.Model) {
|
||||
return 0, fmt.Errorf("modèle de dictée absent : %s (à télécharger dans Paramètres → Dictée)", cfg.Model)
|
||||
}
|
||||
note := ""
|
||||
gpu := cfg.cudaVisibleDevices() != ""
|
||||
bin := whisperServerBin(gpu)
|
||||
if bin == "" && gpu {
|
||||
// Image bâtie sans CUDA : on le DIT, au lieu de laisser croire que le
|
||||
// GPU choisi est utilisé.
|
||||
note = "binaire CUDA absent de cette image — dictée sur le processeur"
|
||||
gpu = false
|
||||
cfg.Device = "cpu"
|
||||
bin = whisperServerBin(false)
|
||||
}
|
||||
if bin == "" {
|
||||
return 0, fmt.Errorf("whisper-server introuvable (image à reconstruire)")
|
||||
}
|
||||
port, err := portLibre()
|
||||
if err != nil {
|
||||
return 0, fmt.Errorf("aucun port libre pour whisper-server : %w", err)
|
||||
}
|
||||
args, err := whisperServerArgs(cfg, port)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
log := &bytes.Buffer{}
|
||||
cmd := exec.Command(bin, args...)
|
||||
cmd.Env = whisperServerEnv(cfg, os.Environ())
|
||||
cmd.Stdout, cmd.Stderr = log, log
|
||||
if err := cmd.Start(); err != nil {
|
||||
return 0, fmt.Errorf("démarrage de whisper-server : %w", err)
|
||||
}
|
||||
wsrv.cmd, wsrv.port, wsrv.cfg = cmd, port, dictateCfgLoad()
|
||||
wsrv.log, wsrv.note, wsrv.depuis = log, note, time.Now()
|
||||
if err := whisperAttendPret(cmd, port, log); err != nil {
|
||||
whisperStopLocked()
|
||||
return 0, err
|
||||
}
|
||||
whisperTouchLocked()
|
||||
return port, nil
|
||||
}
|
||||
|
||||
// whisperAttendPret sonde le serveur jusqu'à ce qu'il réponde. Si le processus
|
||||
// meurt entre-temps, on rapporte COMMENT il est mort : sur une mort par
|
||||
// signal, la dernière ligne du journal est celle d'avant le coup fatal — elle
|
||||
// ressemble à une explication sans en être une (leçon de la PR #18).
|
||||
func whisperAttendPret(cmd *exec.Cmd, port int, log *bytes.Buffer) error {
|
||||
fin := time.Now().Add(whisperReady)
|
||||
url := "http://127.0.0.1:" + strconv.Itoa(port) + "/"
|
||||
mort := make(chan error, 1)
|
||||
go func() { mort <- cmd.Wait() }()
|
||||
for time.Now().Before(fin) {
|
||||
select {
|
||||
case err := <-mort:
|
||||
msg := whisperExitReason(err)
|
||||
if l := lastLine(log.String()); l != "" {
|
||||
msg += " — dernière sortie : " + l
|
||||
}
|
||||
return fmt.Errorf("whisper-server s'est arrêté au démarrage : %s", msg)
|
||||
case <-time.After(200 * time.Millisecond):
|
||||
}
|
||||
req, err := http.NewRequest(http.MethodGet, url, nil)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 2*time.Second)
|
||||
resp, err := http.DefaultClient.Do(req.WithContext(ctx))
|
||||
cancel()
|
||||
if err == nil {
|
||||
_ = resp.Body.Close()
|
||||
return nil
|
||||
}
|
||||
}
|
||||
return fmt.Errorf("whisper-server n'a pas répondu en %s", whisperReady)
|
||||
}
|
||||
|
||||
func whisperTouchLocked() {
|
||||
if wsrv.timer != nil {
|
||||
wsrv.timer.Stop()
|
||||
}
|
||||
wsrv.timer = time.AfterFunc(whisperIdle, whisperShutdown)
|
||||
}
|
||||
|
||||
func whisperStopLocked() {
|
||||
if wsrv.timer != nil {
|
||||
wsrv.timer.Stop()
|
||||
wsrv.timer = nil
|
||||
}
|
||||
if wsrv.cmd != nil && wsrv.cmd.Process != nil {
|
||||
_ = wsrv.cmd.Process.Kill()
|
||||
}
|
||||
wsrv.cmd, wsrv.port, wsrv.log = nil, 0, nil
|
||||
wsrv.cfg, wsrv.note = DictateCfg{}, ""
|
||||
}
|
||||
|
||||
// whisperShutdown arrête le serveur. Appelé par la minuterie d'inactivité et
|
||||
// à chaque changement de réglages — sans ça, le serveur continuerait de
|
||||
// tourner avec l'ancien modèle et le nouveau réglage semblerait sans effet.
|
||||
func whisperShutdown() {
|
||||
wsrvMu.Lock()
|
||||
defer wsrvMu.Unlock()
|
||||
whisperStopLocked()
|
||||
}
|
||||
|
||||
// whisperInfer garantit le serveur puis lui soumet le WAV.
|
||||
func whisperInfer(ctx context.Context, wav []byte) (string, error) {
|
||||
port, err := whisperEnsure()
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
return whisperInferSur(ctx, "http://127.0.0.1:"+strconv.Itoa(port), wav)
|
||||
}
|
||||
|
||||
func whisperEtat() map[string]any {
|
||||
cfg := dictateCfgLoad()
|
||||
wsrvMu.Lock()
|
||||
defer wsrvMu.Unlock()
|
||||
etat := map[string]any{
|
||||
"actif": wsrv.cmd != nil,
|
||||
"modele": cfg.Model,
|
||||
"present": whisperModelPresent(cfg.Model),
|
||||
"device": cfg.Device,
|
||||
"note": wsrv.note,
|
||||
}
|
||||
if wsrv.cmd != nil {
|
||||
etat["depuis_s"] = int(time.Since(wsrv.depuis).Seconds())
|
||||
}
|
||||
return etat
|
||||
}
|
||||
@@ -0,0 +1,188 @@
|
||||
package loki
|
||||
|
||||
import (
|
||||
"context"
|
||||
"net"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"strconv"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func argVal(args []string, flag string) (string, bool) {
|
||||
for i, a := range args {
|
||||
if a == flag && i+1 < len(args) {
|
||||
return args[i+1], true
|
||||
}
|
||||
}
|
||||
return "", false
|
||||
}
|
||||
|
||||
func TestServerArgs(t *testing.T) {
|
||||
testHome(t)
|
||||
c := DictateCfg{Model: "small-q5_1", Lang: "fr", Device: "cpu"}.withDefauts()
|
||||
args, err := whisperServerArgs(c, 8091)
|
||||
if err != nil {
|
||||
t.Fatalf("construction des arguments : %v", err)
|
||||
}
|
||||
if v, ok := argVal(args, "-m"); !ok || !strings.HasSuffix(v, "ggml-small-q5_1.bin") {
|
||||
t.Errorf("-m = %q, doit pointer sur le .bin du modèle choisi", v)
|
||||
}
|
||||
if v, ok := argVal(args, "-l"); !ok || v != "fr" {
|
||||
t.Errorf("-l = %q, attendu fr", v)
|
||||
}
|
||||
if v, ok := argVal(args, "--port"); !ok || v != "8091" {
|
||||
t.Errorf("--port = %q, attendu 8091", v)
|
||||
}
|
||||
// whisper-server n'a aucune authentification : l'exposer sur 0.0.0.0
|
||||
// ouvrirait la transcription à tout le réseau.
|
||||
if v, ok := argVal(args, "--host"); !ok || v != "127.0.0.1" {
|
||||
t.Errorf("--host = %q, attendu 127.0.0.1", v)
|
||||
}
|
||||
}
|
||||
|
||||
func TestServerArgsModeleInconnu(t *testing.T) {
|
||||
testHome(t)
|
||||
if _, err := whisperServerArgs(DictateCfg{Model: "nawak", Lang: "fr", Device: "cpu"}, 8091); err == nil {
|
||||
t.Error("un modèle hors catalogue doit produire une erreur, pas une commande avec -m vide")
|
||||
}
|
||||
}
|
||||
|
||||
func TestServerEnvGPU(t *testing.T) {
|
||||
base := []string{"PATH=/usr/bin", "CUDA_VISIBLE_DEVICES=7", "HOME=/root"}
|
||||
|
||||
env := whisperServerEnv(DictateCfg{Device: "1"}, base)
|
||||
if !contientExact(env, "CUDA_VISIBLE_DEVICES=1") {
|
||||
t.Errorf("env = %v, doit contenir CUDA_VISIBLE_DEVICES=1", env)
|
||||
}
|
||||
// L'ancienne valeur doit être REMPLACÉE, pas doublée : avec deux
|
||||
// occurrences, le gagnant dépend de l'implémentation — le choix de
|
||||
// l'utilisateur ne doit pas dépendre de ça.
|
||||
if n := compte(env, "CUDA_VISIBLE_DEVICES="); n != 1 {
|
||||
t.Errorf("%d occurrences de CUDA_VISIBLE_DEVICES, attendu exactement 1", n)
|
||||
}
|
||||
if !contientExact(env, "PATH=/usr/bin") {
|
||||
t.Error("le reste de l'environnement doit être conservé")
|
||||
}
|
||||
|
||||
env = whisperServerEnv(DictateCfg{Device: "cpu"}, base)
|
||||
if !contientExact(env, "CUDA_VISIBLE_DEVICES=") {
|
||||
t.Errorf("env = %v, en mode CPU la variable doit être posée VIDE (et non retirée) pour masquer tous les GPU", env)
|
||||
}
|
||||
}
|
||||
|
||||
func contientExact(l []string, s string) bool {
|
||||
for _, v := range l {
|
||||
if v == s {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func compte(l []string, prefixe string) int {
|
||||
n := 0
|
||||
for _, v := range l {
|
||||
if strings.HasPrefix(v, prefixe) {
|
||||
n++
|
||||
}
|
||||
}
|
||||
return n
|
||||
}
|
||||
|
||||
func TestPortLibre(t *testing.T) {
|
||||
p, err := portLibre()
|
||||
if err != nil {
|
||||
t.Fatalf("portLibre : %v", err)
|
||||
}
|
||||
if p < 1024 || p > 65535 {
|
||||
t.Errorf("port = %d, hors de la plage utilisable", p)
|
||||
}
|
||||
l, err := net.Listen("tcp", "127.0.0.1:"+strconv.Itoa(p))
|
||||
if err != nil {
|
||||
t.Fatalf("port %d annoncé libre mais inutilisable : %v", p, err)
|
||||
}
|
||||
_ = l.Close()
|
||||
}
|
||||
|
||||
func TestInferParseLaReponse(t *testing.T) {
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
if r.URL.Path != "/inference" {
|
||||
t.Errorf("chemin appelé = %q, attendu /inference", r.URL.Path)
|
||||
}
|
||||
if ct := r.Header.Get("Content-Type"); !strings.HasPrefix(ct, "multipart/form-data") {
|
||||
t.Errorf("Content-Type = %q, whisper-server attend du multipart", ct)
|
||||
}
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
_, _ = w.Write([]byte(`{"text":" bonjour ceci est un test "}`))
|
||||
}))
|
||||
defer srv.Close()
|
||||
|
||||
txt, err := whisperInferSur(context.Background(), srv.URL, []byte("RIFFfaux"))
|
||||
if err != nil {
|
||||
t.Fatalf("whisperInferSur : %v", err)
|
||||
}
|
||||
if txt != "bonjour ceci est un test" {
|
||||
t.Errorf("texte = %q, les espaces de bord doivent être retirés", txt)
|
||||
}
|
||||
}
|
||||
|
||||
// Sur du silence, whisper renvoie ses marqueurs internes. Les laisser passer
|
||||
// les collerait tels quels dans le champ de saisie — c'est ce qui est arrivé
|
||||
// avec [BLANK_AUDIO].
|
||||
func TestInferFiltreLesMarqueurs(t *testing.T) {
|
||||
for _, marqueur := range []string{"[BLANK_AUDIO]", "(silence)", "[SOUND]", " [ Silence ] ", " "} {
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
_, _ = w.Write([]byte(`{"text":"` + marqueur + `"}`))
|
||||
}))
|
||||
txt, err := whisperInferSur(context.Background(), srv.URL, []byte("RIFFfaux"))
|
||||
srv.Close()
|
||||
if err != nil {
|
||||
t.Fatalf("whisperInferSur(%q) : %v", marqueur, err)
|
||||
}
|
||||
if txt != "" {
|
||||
t.Errorf("marqueur %q rendu comme texte %q, attendu vide", marqueur, txt)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestInferErreurHTTP(t *testing.T) {
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
w.WriteHeader(500)
|
||||
_, _ = w.Write([]byte("boom"))
|
||||
}))
|
||||
defer srv.Close()
|
||||
if _, err := whisperInferSur(context.Background(), srv.URL, []byte("RIFFfaux")); err == nil {
|
||||
t.Error("un 500 de whisper-server doit remonter une erreur")
|
||||
}
|
||||
}
|
||||
|
||||
// Un modèle absent doit être annoncé comme tel, pas produire un lancement de
|
||||
// whisper-server voué à mourir sur un fichier introuvable.
|
||||
func TestEnsureModeleAbsent(t *testing.T) {
|
||||
testHome(t)
|
||||
if err := dictateCfgSave(DictateCfg{Model: "small-q5_1", Lang: "fr", Device: "cpu", Reactivity: "moyen"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
_, err := whisperEnsure()
|
||||
if err == nil {
|
||||
t.Fatal("un modèle absent doit produire une erreur")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "absent") {
|
||||
t.Errorf("erreur = %q, elle doit dire que le modèle est absent", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestEtatSansServeur(t *testing.T) {
|
||||
testHome(t)
|
||||
e := whisperEtat()
|
||||
if e["actif"] != false {
|
||||
t.Errorf("actif = %v, attendu false sans serveur lancé", e["actif"])
|
||||
}
|
||||
for _, clef := range []string{"modele", "present", "device", "note"} {
|
||||
if _, ok := e[clef]; !ok {
|
||||
t.Errorf("clé %q manquante — l'UI en a besoin", clef)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -168,8 +168,12 @@ func newWebMux() *http.ServeMux {
|
||||
api := func(path string, h http.HandlerFunc) { mux.HandleFunc(path, requireWebAuth(h)) }
|
||||
api("/api/ping", handlePing)
|
||||
api("/api/status", handleStatus)
|
||||
api("/api/transcribe", handleTranscribe) // dictée vocale (whisper.cpp local)
|
||||
api("/api/service/log", handleServiceLog) // journal du service pour diagnostiquer un modèle qui ne charge pas
|
||||
api("/api/transcribe", handleTranscribe) // dictée vocale (whisper.cpp local)
|
||||
api("/api/dictate/config", handleDictateConfig) // réglages : modèle, langue, matériel, réactivité
|
||||
api("/api/dictate/models", handleDictateModels) // catalogue + présence sur disque
|
||||
api("/api/dictate/models/download", handleDictateDownload) // télécharge un modèle
|
||||
api("/api/dictate/state", handleDictateState) // serveur allumé, modèle chargé, dernière erreur
|
||||
api("/api/service/log", handleServiceLog) // journal du service pour diagnostiquer un modèle qui ne charge pas
|
||||
api("/api/vram", handleVram)
|
||||
api("/api/ram", handleRam)
|
||||
api("/api/config", handleConfigEnv)
|
||||
|
||||
+120
-106
@@ -1,13 +1,17 @@
|
||||
// Dictée vocale — POST /api/transcribe : un WAV (16 kHz mono, encodé côté
|
||||
// navigateur) entre, le texte transcrit sort. La transcription est 100 % locale,
|
||||
// par whisper-cli (whisper.cpp, même famille que llama.cpp), compilé dans
|
||||
// l'image Docker. Le modèle (ggml-small-q5_1, ~190 Mo, multilingue) est
|
||||
// téléchargé au premier usage dans LOKI_HOME/whisper/ — comme le moteur, il ne
|
||||
// gonfle pas l'image et survit aux recréations du conteneur via /data.
|
||||
// par whisper.cpp. Le modèle est téléchargé au premier usage dans
|
||||
// LOKI_HOME/whisper/ — comme le moteur, il ne gonfle pas l'image et survit aux
|
||||
// recréations du conteneur via /data.
|
||||
//
|
||||
// La transcription elle-même passe par whisper-server (dictate_server.go), qui
|
||||
// garde le modèle chargé. L'ancien chemin relançait whisper-cli à chaque
|
||||
// requête, donc rechargeait le modèle à chaque phrase dictée.
|
||||
package loki
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
@@ -15,52 +19,35 @@ import (
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"runtime"
|
||||
"strconv"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
)
|
||||
|
||||
const whisperModelURL = "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-small-q5_1.bin"
|
||||
|
||||
// whisperBin localise whisper-cli : variable d'environnement (posée par le
|
||||
// Dockerfile), sinon le PATH. Vide = dictée indisponible (image trop ancienne,
|
||||
// ou poste de dev sans whisper.cpp) — le handler le dit clairement.
|
||||
func whisperBin() string {
|
||||
if p := os.Getenv("LOKI_WHISPER_BIN"); p != "" {
|
||||
if _, err := os.Stat(p); err == nil {
|
||||
return p
|
||||
}
|
||||
}
|
||||
if p, err := exec.LookPath("whisper-cli"); err == nil {
|
||||
return p
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
func whisperModelPath() string {
|
||||
return filepath.Join(LokiHome(), "whisper", "ggml-small-q5_1.bin")
|
||||
}
|
||||
|
||||
// Téléchargement du modèle : UNE goroutine, progression consultable. Le premier
|
||||
// POST le déclenche et répond 503 {downloading, pct} ; l'UI invite à réessayer.
|
||||
// Téléchargement d'un modèle : UNE goroutine à la fois, progression
|
||||
// consultable. Le premier POST le déclenche et répond 503 {downloading, pct} ;
|
||||
// l'UI invite à réessayer.
|
||||
var wspMu sync.Mutex
|
||||
var wspDl struct {
|
||||
id string
|
||||
running bool
|
||||
pct int
|
||||
err string
|
||||
}
|
||||
|
||||
func whisperStartDownload() {
|
||||
func whisperStartDownload(id string) {
|
||||
wspMu.Lock()
|
||||
defer wspMu.Unlock()
|
||||
if wspDl.running {
|
||||
return
|
||||
}
|
||||
wspDl.running, wspDl.pct, wspDl.err = true, 0, ""
|
||||
if whisperModelURLFor(id) == "" {
|
||||
wspDl.err = "modèle de dictée inconnu : " + id
|
||||
return
|
||||
}
|
||||
wspDl.id, wspDl.running, wspDl.pct, wspDl.err = id, true, 0, ""
|
||||
go func() {
|
||||
err := whisperDownload()
|
||||
err := whisperDownload(id)
|
||||
wspMu.Lock()
|
||||
wspDl.running = false
|
||||
if err != nil {
|
||||
@@ -70,19 +57,24 @@ func whisperStartDownload() {
|
||||
}()
|
||||
}
|
||||
|
||||
func whisperDownload() error {
|
||||
dst := whisperModelPath()
|
||||
func whisperDownload(id string) error {
|
||||
dst := whisperModelPathFor(id)
|
||||
if dst == "" {
|
||||
return fmt.Errorf("modèle de dictée inconnu : %s", id)
|
||||
}
|
||||
if err := os.MkdirAll(filepath.Dir(dst), 0o755); err != nil {
|
||||
return err
|
||||
}
|
||||
resp, err := http.Get(whisperModelURL)
|
||||
resp, err := http.Get(whisperModelURLFor(id))
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
defer func() { _ = resp.Body.Close() }()
|
||||
if resp.StatusCode != 200 {
|
||||
return fmt.Errorf("téléchargement du modèle whisper : HTTP %d", resp.StatusCode)
|
||||
}
|
||||
// Écriture dans un .part renommé à la fin : un téléchargement interrompu
|
||||
// ne doit pas laisser un .bin tronqué que whisperModelPresent croirait bon.
|
||||
tmp := dst + ".part"
|
||||
f, err := os.Create(tmp)
|
||||
if err != nil {
|
||||
@@ -95,8 +87,8 @@ func whisperDownload() error {
|
||||
n, rerr := resp.Body.Read(buf)
|
||||
if n > 0 {
|
||||
if _, werr := f.Write(buf[:n]); werr != nil {
|
||||
f.Close()
|
||||
os.Remove(tmp)
|
||||
_ = f.Close()
|
||||
_ = os.Remove(tmp)
|
||||
return werr
|
||||
}
|
||||
done += int64(n)
|
||||
@@ -110,91 +102,110 @@ func whisperDownload() error {
|
||||
break
|
||||
}
|
||||
if rerr != nil {
|
||||
f.Close()
|
||||
os.Remove(tmp)
|
||||
_ = f.Close()
|
||||
_ = os.Remove(tmp)
|
||||
return rerr
|
||||
}
|
||||
}
|
||||
if err := f.Close(); err != nil {
|
||||
os.Remove(tmp)
|
||||
_ = os.Remove(tmp)
|
||||
return err
|
||||
}
|
||||
return os.Rename(tmp, dst)
|
||||
}
|
||||
|
||||
func handleTranscribe(w http.ResponseWriter, r *http.Request) {
|
||||
bin := whisperBin()
|
||||
if bin == "" {
|
||||
sendJSON(w, 501, map[string]any{"error": "dictée indisponible : whisper-cli absent (image à reconstruire, ou poste sans whisper.cpp)"})
|
||||
return
|
||||
}
|
||||
model := whisperModelPath()
|
||||
if _, err := os.Stat(model); err != nil {
|
||||
wspMu.Lock()
|
||||
derr := wspDl.err
|
||||
pct := wspDl.pct
|
||||
wspMu.Unlock()
|
||||
if derr != "" {
|
||||
// Échec précédent : on le dit ET on relance — un réseau revenu suffit.
|
||||
wspMu.Lock()
|
||||
wspDl.err = ""
|
||||
wspMu.Unlock()
|
||||
whisperStartDownload()
|
||||
sendJSON(w, 503, map[string]any{"downloading": true, "pct": 0, "error": derr})
|
||||
return
|
||||
}
|
||||
whisperStartDownload()
|
||||
sendJSON(w, 503, map[string]any{"downloading": true, "pct": pct})
|
||||
return
|
||||
}
|
||||
// whisperDlEtat : instantané de la progression, pour l'UI et pour les 503.
|
||||
func whisperDlEtat() (id string, running bool, pct int, errMsg string) {
|
||||
wspMu.Lock()
|
||||
defer wspMu.Unlock()
|
||||
return wspDl.id, wspDl.running, wspDl.pct, wspDl.err
|
||||
}
|
||||
|
||||
func handleTranscribe(w http.ResponseWriter, r *http.Request) {
|
||||
audio, err := io.ReadAll(http.MaxBytesReader(w, r.Body, 32<<20))
|
||||
if err != nil || len(audio) < 1000 {
|
||||
sendJSON(w, 400, map[string]any{"error": "audio vide ou trop court"})
|
||||
return
|
||||
}
|
||||
tmp, err := os.CreateTemp("", "loki-dictee-*.wav")
|
||||
if err != nil {
|
||||
sendJSON(w, 500, map[string]any{"error": err.Error()})
|
||||
return
|
||||
}
|
||||
tmpPath := tmp.Name()
|
||||
defer os.Remove(tmpPath)
|
||||
if _, err := tmp.Write(audio); err != nil {
|
||||
tmp.Close()
|
||||
sendJSON(w, 500, map[string]any{"error": err.Error()})
|
||||
return
|
||||
}
|
||||
tmp.Close()
|
||||
|
||||
// -l auto : détection de langue (le français sort en français) ; -nt sans
|
||||
// horodatages ; -np sans bavardage de chargement. CPU seul : une dictée de
|
||||
// quelques dizaines de secondes se transcrit en 1-3 s avec small-q5_1.
|
||||
threads := runtime.NumCPU()
|
||||
if threads > 8 {
|
||||
threads = 8
|
||||
}
|
||||
ctx, cancel := context.WithTimeout(r.Context(), 3*time.Minute)
|
||||
defer cancel()
|
||||
cmd := exec.CommandContext(ctx, bin, "-m", model, "-f", tmpPath, "-l", "auto", "-nt", "-np", "-t", strconv.Itoa(threads))
|
||||
var stderr strings.Builder
|
||||
cmd.Stderr = &stderr
|
||||
out, err := cmd.Output()
|
||||
if err != nil {
|
||||
// Ne garder QUE stderr est trompeur quand le binaire meurt d'un signal :
|
||||
// la dernière ligne écrite avant l'exécution du coup fatal ressemble à
|
||||
// une explication (« AMX is not ready to be used! ») alors que la vraie
|
||||
// cause est la mort brutale. On dit donc toujours comment il a fini —
|
||||
// un SIGILL/SIGSEGV désigne un binaire compilé pour un autre
|
||||
// processeur, pas un problème d'audio.
|
||||
msg := whisperExitReason(err)
|
||||
if last := lastLine(stderr.String()); last != "" {
|
||||
msg += " — dernière sortie : " + last
|
||||
cfg := dictateCfgLoad()
|
||||
if !whisperModelPresent(cfg.Model) {
|
||||
_, _, pct, derr := whisperDlEtat()
|
||||
if derr != "" {
|
||||
// Échec précédent : on le dit ET on relance — un réseau revenu suffit.
|
||||
wspMu.Lock()
|
||||
wspDl.err = ""
|
||||
wspMu.Unlock()
|
||||
whisperStartDownload(cfg.Model)
|
||||
sendJSON(w, 503, map[string]any{"downloading": true, "pct": 0, "error": derr})
|
||||
return
|
||||
}
|
||||
sendJSON(w, 500, map[string]any{"error": "whisper-cli : " + msg})
|
||||
whisperStartDownload(cfg.Model)
|
||||
sendJSON(w, 503, map[string]any{"downloading": true, "pct": pct})
|
||||
return
|
||||
}
|
||||
sendJSON(w, 200, map[string]any{"text": strings.TrimSpace(string(out))})
|
||||
// Le chargement du modèle au premier appel peut être long ; la
|
||||
// transcription elle-même ne l'est pas.
|
||||
ctx, cancel := context.WithTimeout(r.Context(), 5*time.Minute)
|
||||
defer cancel()
|
||||
txt, err := whisperInfer(ctx, audio)
|
||||
if err != nil {
|
||||
sendJSON(w, 500, map[string]any{"error": err.Error()})
|
||||
return
|
||||
}
|
||||
sendJSON(w, 200, map[string]any{"text": txt})
|
||||
}
|
||||
|
||||
// ─── Réglages de la dictée ───────────────────────────────────────────────────
|
||||
|
||||
func handleDictateConfig(w http.ResponseWriter, r *http.Request) {
|
||||
if r.Method == http.MethodPost {
|
||||
var c DictateCfg
|
||||
if err := json.NewDecoder(r.Body).Decode(&c); err != nil {
|
||||
sendJSON(w, 400, map[string]any{"error": "requête illisible"})
|
||||
return
|
||||
}
|
||||
if err := dictateCfgSave(c); err != nil {
|
||||
sendJSON(w, 400, map[string]any{"error": err.Error()})
|
||||
return
|
||||
}
|
||||
// Sans ça, le serveur continuerait de tourner avec l'ancien modèle et
|
||||
// le nouveau réglage semblerait sans effet.
|
||||
whisperShutdown()
|
||||
sendJSON(w, 200, map[string]any{"ok": true})
|
||||
return
|
||||
}
|
||||
sendJSON(w, 200, dictateCfgLoad())
|
||||
}
|
||||
|
||||
func handleDictateModels(w http.ResponseWriter, r *http.Request) {
|
||||
sendJSON(w, 200, whisperCatalogueTrie())
|
||||
}
|
||||
|
||||
func handleDictateDownload(w http.ResponseWriter, r *http.Request) {
|
||||
var req struct {
|
||||
ID string `json:"id"`
|
||||
}
|
||||
if r.Method == http.MethodPost {
|
||||
if err := json.NewDecoder(r.Body).Decode(&req); err != nil {
|
||||
sendJSON(w, 400, map[string]any{"error": "requête illisible"})
|
||||
return
|
||||
}
|
||||
id := strings.TrimSpace(req.ID)
|
||||
if whisperModelURLFor(id) == "" {
|
||||
sendJSON(w, 400, map[string]any{"error": "modèle de dictée inconnu : " + id})
|
||||
return
|
||||
}
|
||||
whisperStartDownload(id)
|
||||
}
|
||||
id, running, pct, errMsg := whisperDlEtat()
|
||||
sendJSON(w, 200, map[string]any{"id": id, "running": running, "pct": pct, "error": errMsg})
|
||||
}
|
||||
|
||||
func handleDictateState(w http.ResponseWriter, r *http.Request) {
|
||||
etat := whisperEtat()
|
||||
id, running, pct, errMsg := whisperDlEtat()
|
||||
etat["dl"] = map[string]any{"id": id, "running": running, "pct": pct, "error": errMsg}
|
||||
sendJSON(w, 200, etat)
|
||||
}
|
||||
|
||||
// whisperExitReason traduit la fin du processus en une phrase utilisable.
|
||||
@@ -204,6 +215,9 @@ func handleTranscribe(w http.ResponseWriter, r *http.Request) {
|
||||
func whisperExitReason(err error) string {
|
||||
var ee *exec.ExitError
|
||||
if !errors.As(err, &ee) {
|
||||
if err == nil {
|
||||
return "arrêt sans erreur"
|
||||
}
|
||||
return err.Error()
|
||||
}
|
||||
st := ee.ProcessState.String()
|
||||
@@ -213,7 +227,7 @@ func whisperExitReason(err error) string {
|
||||
return st
|
||||
}
|
||||
|
||||
// lastLine : la dernière ligne non vide. whisper-cli bavarde beaucoup avant de
|
||||
// lastLine : la dernière ligne non vide. whisper bavarde beaucoup avant de
|
||||
// tomber ; seule la fin renseigne, et un pavé ne tient pas dans un bandeau.
|
||||
func lastLine(s string) string {
|
||||
lines := strings.Split(strings.TrimSpace(s), "\n")
|
||||
|
||||
@@ -6,30 +6,115 @@ import (
|
||||
"testing"
|
||||
)
|
||||
|
||||
// La dictée est morte en production parce que whisper-cli avait été compilé
|
||||
// avec -march=native, donc pour le processeur du runner GitHub (AVX-512, AMX)
|
||||
// et pas pour la machine qui fait tourner l'image : SIGILL en pleine
|
||||
// transcription. Le drapeau qui l'évite ne se voit pas à l'exécution — rien ne
|
||||
// le rappelle au prochain qui touchera cette étape de build. Ce test le fait.
|
||||
// etapesWhisperbuild découpe le Dockerfile en étapes « FROM … AS whisperbuild* »
|
||||
// et rend le texte de chacune.
|
||||
func etapesWhisperbuild(t *testing.T) map[string]string {
|
||||
t.Helper()
|
||||
b, err := os.ReadFile("../../Dockerfile")
|
||||
if err != nil {
|
||||
t.Fatalf("lecture du Dockerfile : %v", err)
|
||||
}
|
||||
out := map[string]string{}
|
||||
for _, bloc := range strings.Split(string(b), "\nFROM ") {
|
||||
entete, reste, ok := strings.Cut(bloc, "\n")
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
_, nom, ok := strings.Cut(entete, " AS ")
|
||||
if !ok || !strings.HasPrefix(strings.TrimSpace(nom), "whisperbuild") {
|
||||
continue
|
||||
}
|
||||
out[strings.TrimSpace(nom)] = entete + "\n" + reste
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// argWhisperFlags rend la valeur de l'ARG WHISPER_CMAKE_FLAGS, continuations
|
||||
// de ligne comprises. Vide si l'ARG n'existe pas : les étapes devront alors
|
||||
// porter les drapeaux en clair, et le test le vérifiera.
|
||||
func argWhisperFlags(t *testing.T) string {
|
||||
t.Helper()
|
||||
b, err := os.ReadFile("../../Dockerfile")
|
||||
if err != nil {
|
||||
t.Fatalf("lecture du Dockerfile : %v", err)
|
||||
}
|
||||
_, apres, ok := strings.Cut(string(b), "ARG WHISPER_CMAKE_FLAGS=")
|
||||
if !ok {
|
||||
return ""
|
||||
}
|
||||
var val strings.Builder
|
||||
for _, ligne := range strings.Split(apres, "\n") {
|
||||
val.WriteString(" " + ligne)
|
||||
if !strings.HasSuffix(strings.TrimSpace(ligne), "\\") {
|
||||
break
|
||||
}
|
||||
}
|
||||
return val.String()
|
||||
}
|
||||
|
||||
// La dictée est morte en production parce que whisper avait été compilé avec
|
||||
// -march=native, donc pour le processeur du runner GitHub (AVX-512, AMX) et pas
|
||||
// pour la machine qui fait tourner l'image : SIGILL en pleine transcription.
|
||||
// Les drapeaux qui l'évitent ne se voient pas à l'exécution — rien ne les
|
||||
// rappelle au prochain qui touchera cette étape de build. Ce test le fait, sur
|
||||
// CHAQUE étape, y compris celle ajoutée pour CUDA.
|
||||
func TestDockerfileWhisperNonNatif(t *testing.T) {
|
||||
etapes := etapesWhisperbuild(t)
|
||||
if len(etapes) < 2 {
|
||||
t.Fatalf("%d étape(s) whisperbuild, 2 attendues (CPU et CUDA) : %v", len(etapes), clefs(etapes))
|
||||
}
|
||||
// Les drapeaux communs sont factorisés dans WHISPER_CMAKE_FLAGS. Une étape
|
||||
// est conforme si elle les porte en clair OU si elle référence cet ARG —
|
||||
// ce qui compte est qu'ils atteignent cmake, pas qu'ils soient recopiés.
|
||||
commun := argWhisperFlags(t)
|
||||
for nom, txt := range etapes {
|
||||
effectif := txt
|
||||
if strings.Contains(txt, "${WHISPER_CMAKE_FLAGS}") {
|
||||
effectif += " " + commun
|
||||
}
|
||||
for _, drapeau := range []string{"-DGGML_NATIVE=OFF", "-DGGML_AMX_TILE=OFF", "-DGGML_AVX512=OFF"} {
|
||||
if !strings.Contains(effectif, drapeau) {
|
||||
t.Errorf("étape %s : %s absent — le binaire sera compilé pour le processeur du runner et mourra d'un SIGILL ailleurs", nom, drapeau)
|
||||
}
|
||||
}
|
||||
if !strings.Contains(txt, "whisper-server") {
|
||||
t.Errorf("étape %s : ne construit pas la cible whisper-server", nom)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Deux binaires, pas un : sur une image runtime bâtie sans CUDA, un binaire lié
|
||||
// à CUDA ne démarre pas du tout — l'éditeur de liens échoue avant la première
|
||||
// instruction, donc aucun repli n'est possible depuis le programme.
|
||||
func TestDockerfileDeuxBinairesWhisper(t *testing.T) {
|
||||
b, err := os.ReadFile("../../Dockerfile")
|
||||
if err != nil {
|
||||
t.Fatalf("lecture du Dockerfile : %v", err)
|
||||
}
|
||||
src := string(b)
|
||||
_, apres, ok := strings.Cut(src, "AS whisperbuild")
|
||||
if !ok {
|
||||
t.Fatal("étape whisperbuild introuvable dans le Dockerfile")
|
||||
}
|
||||
// L'étape suivante commence au FROM d'après : ne pas déborder dessus.
|
||||
if fin := strings.Index(apres, "\nFROM "); fin >= 0 {
|
||||
apres = apres[:fin]
|
||||
}
|
||||
for _, drapeau := range []string{"-DGGML_NATIVE=OFF", "-DGGML_AMX_TILE=OFF"} {
|
||||
if !strings.Contains(apres, drapeau) {
|
||||
t.Errorf("l'étape whisperbuild ne passe plus %s : le binaire sera compilé pour le processeur du runner et mourra d'un SIGILL ailleurs", drapeau)
|
||||
var avecCuda bool
|
||||
for nom, txt := range etapesWhisperbuild(t) {
|
||||
if strings.Contains(txt, "-DGGML_CUDA=ON") {
|
||||
avecCuda = true
|
||||
_ = nom
|
||||
}
|
||||
}
|
||||
if !avecCuda {
|
||||
t.Error("aucune étape whisperbuild ne passe -DGGML_CUDA=ON : la dictée ne pourra jamais utiliser le GPU")
|
||||
}
|
||||
for _, bin := range []string{"whisper-server-cpu", "whisper-server-cuda"} {
|
||||
if !strings.Contains(src, "/usr/local/bin/"+bin) {
|
||||
t.Errorf("le runtime ne reçoit pas %s — dictate_server.go le cherche à cet emplacement", bin)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func clefs(m map[string]string) []string {
|
||||
out := make([]string, 0, len(m))
|
||||
for k := range m {
|
||||
out = append(out, k)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func TestLastLine(t *testing.T) {
|
||||
|
||||
Reference in new issue
Block a user