Arrêt du moteur : attendre tout l'arbre, pas seulement le chef

En conteneur, userSvcStop n'attendait que le processus `loki serve`. Avec
Strata, le moteur qui tient la VRAM (cache d'experts, ~9 Go) est un autre
processus qui finit de s'arrêter après le serveur Python : la relance
suivante démarrait llama-server sur une carte encore pleine (« cudaMalloc
failed: out of memory » à l'encodeur de vision) et Strata mesurait la RAM
disponible avec l'ancien moteur encore chargé (mode mmap choisi à tort).

stopProcTree relève l'arbre (session, groupe et descendants) avant tout
signal, SIGTERM puis SIGKILL aux survivants, et ne revient qu'une fois
tout parti.

Diagnostic : une pile d'appels citant load_model ne passe plus pour une
nouvelle tentative (la cause mémoire était perdue), et le chat donne
l'échec de chargement au lieu de « il démarre, réessaie ».

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_014wfx8uW1un1WwWJ6xzAbkV
This commit is contained in:
Claude committed 2026-10-08 07:18:53 +00:00
1 parent 7d614324a0
commit 393004cb06
6 files changed
+295 -13

No files matched your search

+15
View File
@@ -362,3 +362,18 @@ func TestStrataApplySettingsExperts(t *testing.T) {
t.Errorf("vide doit laisser le choix en place : %q", out) t.Errorf("vide doit laisser le choix en place : %q", out)
} }
} }
// Vécu : bascule de Strata vers llama-server, l'encodeur de vision ne trouve
// plus de VRAM. La pile d'appels cite « load_model » : elle ne doit pas passer
// pour une nouvelle tentative qui effacerait la cause, écrite juste avant.
func TestModelLoadErrorOOMBeforeBacktrace(t *testing.T) {
log := "0.00.247.285 I srv load_model: loading model '/data/models/Swift.gguf'\n" +
"4.47.957.764 E ggml_backend_cuda_buffer_type_alloc_buffer: allocating 884.62 MiB on device 0: cudaMalloc failed: out of memory\n" +
"/app/ggml/src/ggml-backend.cpp:328: GGML_ASSERT(buffer) failed\n" +
"/data/engine/libmtmd.so.0(mtmd_init_from_file+0x31)[0x14bd4fea4c01]\n" +
"/data/engine/libllama-server-impl.so(_ZN19server_context_impl10load_modelER13common_params+0x7ff)[0x14bd50b9f1ff]\n" +
"/lib/x86_64-linux-gnu/libc.so.6(__libc_start_main+0x8b)[0x14bd5059a28b]\n"
if got := modelLoadErrorFrom(log); !strings.Contains(got, "mémoire GPU insuffisante") {
t.Errorf("cause mémoire perdue derrière la pile d'appels : %q", got)
}
}
+6
View File
@@ -1069,6 +1069,12 @@ func friendlyLLMError(err error) error {
} }
func errEngineDown() error { func errEngineDown() error {
// Un refus de connexion ne veut pas toujours dire « il charge encore » : si
// le journal montre que le dernier lancement a échoué, c'est cela qu'il faut
// dire — réessayer dans quelques secondes n'y changera rien.
if msg := modelLoadError(); msg != "" {
return fmt.Errorf("⚠️ %s", msg)
}
return fmt.Errorf("⚠️ Le moteur (llama-server) ne répond pas sur le port %d. Il est probablement en train de démarrer ou de charger le modèle — réessaie dans quelques secondes.", LLMPort()) return fmt.Errorf("⚠️ Le moteur (llama-server) ne répond pas sur le port %d. Il est probablement en train de démarrer ou de charger le modèle — réessaie dans quelques secondes.", LLMPort())
} }
+166
View File
@@ -0,0 +1,166 @@
//go:build linux
package loki
import (
"os"
"strconv"
"strings"
"syscall"
"time"
)
// procRef : un processus repéré par son PID ET son instant de démarrage — un
// PID recyclé pendant l'attente ne passe pas pour le même processus.
type procRef struct {
pid int
start string
}
// procStat : les champs de /proc/<pid>/stat utiles ici.
type procStat struct {
state string
ppid, pgrp, sid int
start string
}
// readProcStat lit /proc/<pid>/stat. Le nom (champ 2, entre parenthèses) peut
// contenir des blancs et des parenthèses : on repart de la DERNIÈRE fermante.
func readProcStat(pid int) (procStat, bool) {
b, err := os.ReadFile("/proc/" + strconv.Itoa(pid) + "/stat")
if err != nil {
return procStat{}, false
}
return parseProcStat(string(b))
}
func parseProcStat(s string) (procStat, bool) {
i := strings.LastIndexByte(s, ')')
if i < 0 {
return procStat{}, false
}
// f[0] est le champ 3 (état), f[1] ppid, f[2] groupe, f[3] session, f[19]
// l'instant de démarrage (champ 22).
f := strings.Fields(s[i+1:])
if len(f) < 20 {
return procStat{}, false
}
ppid, _ := strconv.Atoi(f[1])
pgrp, _ := strconv.Atoi(f[2])
sid, _ := strconv.Atoi(f[3])
return procStat{state: f[0], ppid: ppid, pgrp: pgrp, sid: sid, start: f[19]}, true
}
// procTable : tous les processus vivants (les zombies ont fini de tourner : leur
// mémoire, VRAM comprise, est déjà rendue).
func procTable() map[int]procStat {
ents, err := os.ReadDir("/proc")
if err != nil {
return nil
}
t := make(map[int]procStat, len(ents))
for _, e := range ents {
pid, err := strconv.Atoi(e.Name())
if err != nil {
continue
}
if st, ok := readProcStat(pid); ok && st.state != "Z" {
t[pid] = st
}
}
return t
}
// procTreeOf : le chef root et tout ce qui en descend — sa session et son
// groupe (Setsid au lancement du service), plus les descendants par la
// filiation, pour un enfant qui aurait pris sa propre session (le serveur
// Python de Strata lance son moteur dans un processus à part). À relever AVANT
// d'envoyer le moindre signal : une fois le chef mort, ses orphelins sont
// rattachés à PID 1 et la filiation est perdue.
func procTreeOf(t map[int]procStat, root, self int) []procRef {
in := map[int]bool{}
for pid, st := range t {
if pid != self && (pid == root || st.sid == root || st.pgrp == root) {
in[pid] = true
}
}
// Fermeture par la filiation, jusqu'à ce que plus rien ne s'ajoute.
for grew := true; grew; {
grew = false
for pid, st := range t {
if !in[pid] && pid != self && in[st.ppid] {
in[pid] = true
grew = true
}
}
}
refs := make([]procRef, 0, len(in))
for pid := range in {
refs = append(refs, procRef{pid: pid, start: t[pid].start})
}
return refs
}
// procRefAlive : le processus tourne encore (ni disparu, ni zombie, ni
// remplacé par un autre sous le même PID).
func procRefAlive(r procRef) bool {
st, ok := readProcStat(r.pid)
return ok && st.state != "Z" && st.start == r.start
}
func procRefsAlive(refs []procRef) []procRef {
var alive []procRef
for _, r := range refs {
if procRefAlive(r) {
alive = append(alive, r)
}
}
return alive
}
// stopProcTree arrête le service lancé avec Setsid (pid = chef de session) et
// N'EN REVIENT QU'UNE FOIS TOUT L'ARBRE PARTI.
//
// Attendre le seul chef ne suffit pas : avec Strata, le chef est le serveur
// Python, et le moteur qui tient la VRAM (cache d'experts, plusieurs Go) est un
// autre processus qui finit de s'arrêter après lui. La relance qui suivait
// 500 ms plus tard démarrait alors llama-server sur une carte encore pleine —
// « cudaMalloc failed: out of memory » sur un modèle qui tient pourtant — et
// Strata mesurait la RAM disponible avec l'ancien moteur encore chargé.
//
// SIGTERM à tous, grace pour finir proprement, SIGKILL aux survivants, puis
// settle au plus pour que le noyau les ait vraiment retirés (un processus
// coincé dans une lecture disque ne meurt qu'à la sortie de l'appel système).
// Renvoie ceux qui tournent encore au-delà (rien à faire de plus, mais le dire).
func stopProcTree(pid int, grace, settle time.Duration) []procRef {
refs := procTreeOf(procTable(), pid, os.Getpid())
if len(refs) == 0 {
return nil
}
if err := syscall.Kill(-pid, syscall.SIGTERM); err != nil {
_ = syscall.Kill(pid, syscall.SIGTERM)
}
for _, r := range refs {
_ = syscall.Kill(r.pid, syscall.SIGTERM)
}
if left := waitProcRefs(refs, grace); len(left) == 0 {
return nil
}
_ = syscall.Kill(-pid, syscall.SIGKILL)
for _, r := range procRefsAlive(refs) {
_ = syscall.Kill(r.pid, syscall.SIGKILL)
}
return waitProcRefs(refs, settle)
}
// waitProcRefs attend que tous aient disparu, au plus wait. Renvoie les survivants.
func waitProcRefs(refs []procRef, wait time.Duration) []procRef {
deadline := time.Now().Add(wait)
for {
left := procRefsAlive(refs)
if len(left) == 0 || time.Now().After(deadline) {
return left
}
time.Sleep(100 * time.Millisecond)
}
}
+83
View File
@@ -0,0 +1,83 @@
//go:build linux
package loki
import (
"os"
"os/exec"
"strconv"
"strings"
"syscall"
"testing"
"time"
)
func TestParseProcStat(t *testing.T) {
st, ok := parseProcStat("4242 (python3 (x) y) S 4200 4100 4000 0 -1 4194560 0 0 0 0 0 0 0 0 20 0 1 0 987654 0 0")
if !ok || st.state != "S" || st.ppid != 4200 || st.pgrp != 4100 || st.sid != 4000 || st.start != "987654" {
t.Fatalf("lecture fausse : %+v ok=%v", st, ok)
}
if _, ok := parseProcStat("4242 (tronqué"); ok {
t.Error("ligne sans parenthèse fermante acceptée")
}
}
func TestProcTreeOf(t *testing.T) {
tab := map[int]procStat{
100: {ppid: 1, pgrp: 100, sid: 100, start: "a"}, // chef (Setsid)
101: {ppid: 100, pgrp: 100, sid: 100, start: "b"},
200: {ppid: 101, pgrp: 200, sid: 200, start: "c"}, // moteur dans sa propre session
201: {ppid: 200, pgrp: 200, sid: 200, start: "d"},
300: {ppid: 1, pgrp: 300, sid: 300, start: "e"}, // étranger
400: {ppid: 1, pgrp: 100, sid: 100, start: "f"}, // orphelin rattaché à PID 1
500: {ppid: 100, pgrp: 100, sid: 100, start: "g"}, // nous-mêmes
}
got := map[int]bool{}
for _, r := range procTreeOf(tab, 100, 500) {
got[r.pid] = true
}
for _, pid := range []int{100, 101, 200, 201, 400} {
if !got[pid] {
t.Errorf("PID %d oublié", pid)
}
}
if got[300] || got[500] {
t.Errorf("hors de l'arbre pris : %v", got)
}
}
// Le chef meurt vite, son enfant (dans une autre session, comme le moteur de
// Strata) traîne à s'arrêter : stopProcTree ne revient qu'une fois tout parti.
func TestStopProcTreeWaitsForChildren(t *testing.T) {
// Le chef lance un enfant qui ignore SIGTERM dans sa propre session, écrit
// son PID, puis attend.
pidFile := t.TempDir() + "/child"
script := `setsid sh -c 'trap "" TERM; echo $$ > ` + pidFile + `; while :; do sleep 0.1; done' & wait`
cmd := exec.Command("sh", "-c", script)
cmd.SysProcAttr = &syscall.SysProcAttr{Setsid: true}
if err := cmd.Start(); err != nil {
t.Skip("sh indisponible :", err)
}
go cmd.Wait() // pas de zombie pour le chef
var child int
for i := 0; i < 50 && child == 0; i++ {
time.Sleep(50 * time.Millisecond)
if b, err := os.ReadFile(pidFile); err == nil {
child, _ = strconv.Atoi(strings.TrimSpace(string(b)))
}
}
if child == 0 {
t.Fatal("l'enfant n'a pas démarré")
}
st, ok := readProcStat(child)
if !ok || st.sid == cmd.Process.Pid {
t.Fatalf("l'enfant devait avoir sa propre session : %+v", st)
}
ref := procRef{pid: child, start: st.start}
if left := stopProcTree(cmd.Process.Pid, 300*time.Millisecond, 5*time.Second); len(left) != 0 {
t.Fatalf("survivants : %v", left)
}
if procRefAlive(ref) {
t.Fatal("l'enfant tourne encore au retour de stopProcTree")
}
}
+12 -12
View File
@@ -125,25 +125,25 @@ func userCheckStarted(pid int) error {
func userSvcStop(verbose bool) error { func userSvcStop(verbose bool) error {
pid := readServicePID() pid := readServicePID()
if pid <= 0 || !processAlive(pid) { // Le chef parti, ses enfants peuvent tourner encore (moteur de Strata) :
// on ne conclut « rien à arrêter » que si l'arbre entier est vide.
if pid <= 0 || len(procTreeOf(procTable(), pid, os.Getpid())) == 0 {
_ = os.Remove(pidFilePath()) _ = os.Remove(pidFilePath())
if verbose { if verbose {
fmt.Println(yellow("[info]") + " aucun service en cours d'exécution") fmt.Println(yellow("[info]") + " aucun service en cours d'exécution")
} }
return nil return nil
} }
// Setsid a fait de l'enfant un chef de groupe : le PID négatif vise le // Setsid a fait de l'enfant un chef de session : tout ce qu'il a lancé
// groupe entier, donc llama-server s'arrête avec lui. // (llama-server, serveur et moteur de Strata) s'arrête avec lui, et on
if err := syscall.Kill(-pid, syscall.SIGTERM); err != nil { // attend qu'il soit VRAIMENT parti — la relance qui suit en a besoin pour
_ = syscall.Kill(pid, syscall.SIGTERM) // trouver la VRAM et la RAM libres.
} left := stopProcTree(pid, 4*time.Second, 20*time.Second)
for i := 0; i < 40 && processAlive(pid); i++ {
time.Sleep(100 * time.Millisecond)
}
if processAlive(pid) {
_ = syscall.Kill(-pid, syscall.SIGKILL)
}
_ = os.Remove(pidFilePath()) _ = os.Remove(pidFilePath())
if len(left) > 0 {
fmt.Fprintf(os.Stderr, "%s %d processus du moteur toujours là après SIGKILL (PID %d…) : "+
"la mémoire du GPU peut ne pas être encore rendue\n", yellow("[attention]"), len(left), left[0].pid)
}
if verbose { if verbose {
fmt.Println(green("[ok]") + " arrêté") fmt.Println(green("[ok]") + " arrêté")
} }
+13 -1
View File
@@ -107,6 +107,13 @@ func modelLoadErrorFrom(log string) string {
// On ne considère que ce qui suit la DERNIÈRE tentative de chargement. // On ne considère que ce qui suit la DERNIÈRE tentative de chargement.
start := 0 start := 0
for i, l := range lines { for i, l := range lines {
// Une pile d'appels cite les fonctions traversées
// (…server_context_impl10load_model…) : ce n'est pas une nouvelle
// tentative, et repartir de là perdait la cause écrite juste avant
// (« cudaMalloc failed: out of memory » devenait « a planté »).
if isBacktraceFrame(l) {
continue
}
// Un lancement refusé par LOAD_GUARD (backend_serve_moe.go) s'arrête // Un lancement refusé par LOAD_GUARD (backend_serve_moe.go) s'arrête
// avant tout chargement : c'est lui, la dernière tentative. // avant tout chargement : c'est lui, la dernière tentative.
if strings.Contains(l, "loading model") || strings.Contains(l, "load_model") || strings.Contains(l, loadGuardMarker) || if strings.Contains(l, "loading model") || strings.Contains(l, "load_model") || strings.Contains(l, loadGuardMarker) ||
@@ -159,7 +166,8 @@ func modelLoadErrorFrom(log string) string {
strings.Contains(low, "cudamalloc failed"), strings.Contains(low, "cudamalloc failed"),
strings.Contains(low, "failed to allocate"): strings.Contains(low, "failed to allocate"):
// Prioritaire sur l'échec générique : la cause mémoire doit gagner. // Prioritaire sur l'échec générique : la cause mémoire doit gagner.
reason = "mémoire GPU insuffisante (VRAM) — réduis le contexte (CTX) ou prends une quantification plus petite" reason = "mémoire GPU insuffisante (VRAM) — réduis le contexte (CTX) ou prends une quantification plus petite, " +
"et vérifie qu'aucun autre programme n'occupe la carte (nvidia-smi)"
case strings.Contains(low, "ggml_cuda error"), strings.Contains(low, "cuda error"): case strings.Contains(low, "ggml_cuda error"), strings.Contains(low, "cuda error"):
if reason == "" { if reason == "" {
reason = "erreur CUDA — GPU indisponible ? (après un redémarrage du serveur : docker restart loki)" reason = "erreur CUDA — GPU indisponible ? (après un redémarrage du serveur : docker restart loki)"
@@ -196,6 +204,10 @@ func modelLoadErrorFrom(log string) string {
} }
} }
// isBacktraceFrame : une ligne de pile d'appels de glibc
// (« /chemin/lib.so(symbole+0x1c)[0x14bd…] »).
func isBacktraceFrame(l string) bool { return strings.Contains(l, ")[0x") }
// handleServiceLog (GET /api/service/log) : dernières lignes du journal du // handleServiceLog (GET /api/service/log) : dernières lignes du journal du
// service, pour que l'UI puisse montrer POURQUOI le modèle ne se charge pas // service, pour que l'UI puisse montrer POURQUOI le modèle ne se charge pas
// (fichier .gguf illisible, VRAM insuffisante, bibliothèque manquante…) au lieu // (fichier .gguf illisible, VRAM insuffisante, bibliothèque manquante…) au lieu