From 393004cb06e0d52daacfffac0af3889a3ab01178 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 8 Oct 2026 07:18:53 +0000 Subject: [PATCH] =?UTF-8?q?Arr=C3=AAt=20du=20moteur=20:=20attendre=20tout?= =?UTF-8?q?=20l'arbre,=20pas=20seulement=20le=20chef?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit En conteneur, userSvcStop n'attendait que le processus `loki serve`. Avec Strata, le moteur qui tient la VRAM (cache d'experts, ~9 Go) est un autre processus qui finit de s'arrêter après le serveur Python : la relance suivante démarrait llama-server sur une carte encore pleine (« cudaMalloc failed: out of memory » à l'encodeur de vision) et Strata mesurait la RAM disponible avec l'ancien moteur encore chargé (mode mmap choisi à tort). stopProcTree relève l'arbre (session, groupe et descendants) avant tout signal, SIGTERM puis SIGKILL aux survivants, et ne revient qu'une fois tout parti. Diagnostic : une pile d'appels citant load_model ne passe plus pour une nouvelle tentative (la cause mémoire était perdue), et le chat donne l'échec de chargement au lieu de « il démarre, réessaie ». Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_014wfx8uW1un1WwWJ6xzAbkV --- internal/loki/backend_strata_test.go | 15 ++ internal/loki/llm_client.go | 6 + internal/loki/sys_proctree_linux.go | 166 +++++++++++++++++++++++ internal/loki/sys_proctree_linux_test.go | 83 ++++++++++++ internal/loki/sys_service_container.go | 24 ++-- internal/loki/web_api.go | 14 +- 6 files changed, 295 insertions(+), 13 deletions(-) create mode 100644 internal/loki/sys_proctree_linux.go create mode 100644 internal/loki/sys_proctree_linux_test.go diff --git a/internal/loki/backend_strata_test.go b/internal/loki/backend_strata_test.go index dc7545f..2e39ba7 100644 --- a/internal/loki/backend_strata_test.go +++ b/internal/loki/backend_strata_test.go @@ -362,3 +362,18 @@ func TestStrataApplySettingsExperts(t *testing.T) { t.Errorf("vide doit laisser le choix en place : %q", out) } } + +// Vécu : bascule de Strata vers llama-server, l'encodeur de vision ne trouve +// plus de VRAM. La pile d'appels cite « load_model » : elle ne doit pas passer +// pour une nouvelle tentative qui effacerait la cause, écrite juste avant. +func TestModelLoadErrorOOMBeforeBacktrace(t *testing.T) { + log := "0.00.247.285 I srv load_model: loading model '/data/models/Swift.gguf'\n" + + "4.47.957.764 E ggml_backend_cuda_buffer_type_alloc_buffer: allocating 884.62 MiB on device 0: cudaMalloc failed: out of memory\n" + + "/app/ggml/src/ggml-backend.cpp:328: GGML_ASSERT(buffer) failed\n" + + "/data/engine/libmtmd.so.0(mtmd_init_from_file+0x31)[0x14bd4fea4c01]\n" + + "/data/engine/libllama-server-impl.so(_ZN19server_context_impl10load_modelER13common_params+0x7ff)[0x14bd50b9f1ff]\n" + + "/lib/x86_64-linux-gnu/libc.so.6(__libc_start_main+0x8b)[0x14bd5059a28b]\n" + if got := modelLoadErrorFrom(log); !strings.Contains(got, "mémoire GPU insuffisante") { + t.Errorf("cause mémoire perdue derrière la pile d'appels : %q", got) + } +} diff --git a/internal/loki/llm_client.go b/internal/loki/llm_client.go index 8bacd77..6faef72 100644 --- a/internal/loki/llm_client.go +++ b/internal/loki/llm_client.go @@ -1069,6 +1069,12 @@ func friendlyLLMError(err error) error { } func errEngineDown() error { + // Un refus de connexion ne veut pas toujours dire « il charge encore » : si + // le journal montre que le dernier lancement a échoué, c'est cela qu'il faut + // dire — réessayer dans quelques secondes n'y changera rien. + if msg := modelLoadError(); msg != "" { + return fmt.Errorf("⚠️ %s", msg) + } return fmt.Errorf("⚠️ Le moteur (llama-server) ne répond pas sur le port %d. Il est probablement en train de démarrer ou de charger le modèle — réessaie dans quelques secondes.", LLMPort()) } diff --git a/internal/loki/sys_proctree_linux.go b/internal/loki/sys_proctree_linux.go new file mode 100644 index 0000000..4368a90 --- /dev/null +++ b/internal/loki/sys_proctree_linux.go @@ -0,0 +1,166 @@ +//go:build linux + +package loki + +import ( + "os" + "strconv" + "strings" + "syscall" + "time" +) + +// procRef : un processus repéré par son PID ET son instant de démarrage — un +// PID recyclé pendant l'attente ne passe pas pour le même processus. +type procRef struct { + pid int + start string +} + +// procStat : les champs de /proc//stat utiles ici. +type procStat struct { + state string + ppid, pgrp, sid int + start string +} + +// readProcStat lit /proc//stat. Le nom (champ 2, entre parenthèses) peut +// contenir des blancs et des parenthèses : on repart de la DERNIÈRE fermante. +func readProcStat(pid int) (procStat, bool) { + b, err := os.ReadFile("/proc/" + strconv.Itoa(pid) + "/stat") + if err != nil { + return procStat{}, false + } + return parseProcStat(string(b)) +} + +func parseProcStat(s string) (procStat, bool) { + i := strings.LastIndexByte(s, ')') + if i < 0 { + return procStat{}, false + } + // f[0] est le champ 3 (état), f[1] ppid, f[2] groupe, f[3] session, f[19] + // l'instant de démarrage (champ 22). + f := strings.Fields(s[i+1:]) + if len(f) < 20 { + return procStat{}, false + } + ppid, _ := strconv.Atoi(f[1]) + pgrp, _ := strconv.Atoi(f[2]) + sid, _ := strconv.Atoi(f[3]) + return procStat{state: f[0], ppid: ppid, pgrp: pgrp, sid: sid, start: f[19]}, true +} + +// procTable : tous les processus vivants (les zombies ont fini de tourner : leur +// mémoire, VRAM comprise, est déjà rendue). +func procTable() map[int]procStat { + ents, err := os.ReadDir("/proc") + if err != nil { + return nil + } + t := make(map[int]procStat, len(ents)) + for _, e := range ents { + pid, err := strconv.Atoi(e.Name()) + if err != nil { + continue + } + if st, ok := readProcStat(pid); ok && st.state != "Z" { + t[pid] = st + } + } + return t +} + +// procTreeOf : le chef root et tout ce qui en descend — sa session et son +// groupe (Setsid au lancement du service), plus les descendants par la +// filiation, pour un enfant qui aurait pris sa propre session (le serveur +// Python de Strata lance son moteur dans un processus à part). À relever AVANT +// d'envoyer le moindre signal : une fois le chef mort, ses orphelins sont +// rattachés à PID 1 et la filiation est perdue. +func procTreeOf(t map[int]procStat, root, self int) []procRef { + in := map[int]bool{} + for pid, st := range t { + if pid != self && (pid == root || st.sid == root || st.pgrp == root) { + in[pid] = true + } + } + // Fermeture par la filiation, jusqu'à ce que plus rien ne s'ajoute. + for grew := true; grew; { + grew = false + for pid, st := range t { + if !in[pid] && pid != self && in[st.ppid] { + in[pid] = true + grew = true + } + } + } + refs := make([]procRef, 0, len(in)) + for pid := range in { + refs = append(refs, procRef{pid: pid, start: t[pid].start}) + } + return refs +} + +// procRefAlive : le processus tourne encore (ni disparu, ni zombie, ni +// remplacé par un autre sous le même PID). +func procRefAlive(r procRef) bool { + st, ok := readProcStat(r.pid) + return ok && st.state != "Z" && st.start == r.start +} + +func procRefsAlive(refs []procRef) []procRef { + var alive []procRef + for _, r := range refs { + if procRefAlive(r) { + alive = append(alive, r) + } + } + return alive +} + +// stopProcTree arrête le service lancé avec Setsid (pid = chef de session) et +// N'EN REVIENT QU'UNE FOIS TOUT L'ARBRE PARTI. +// +// Attendre le seul chef ne suffit pas : avec Strata, le chef est le serveur +// Python, et le moteur qui tient la VRAM (cache d'experts, plusieurs Go) est un +// autre processus qui finit de s'arrêter après lui. La relance qui suivait +// 500 ms plus tard démarrait alors llama-server sur une carte encore pleine — +// « cudaMalloc failed: out of memory » sur un modèle qui tient pourtant — et +// Strata mesurait la RAM disponible avec l'ancien moteur encore chargé. +// +// SIGTERM à tous, grace pour finir proprement, SIGKILL aux survivants, puis +// settle au plus pour que le noyau les ait vraiment retirés (un processus +// coincé dans une lecture disque ne meurt qu'à la sortie de l'appel système). +// Renvoie ceux qui tournent encore au-delà (rien à faire de plus, mais le dire). +func stopProcTree(pid int, grace, settle time.Duration) []procRef { + refs := procTreeOf(procTable(), pid, os.Getpid()) + if len(refs) == 0 { + return nil + } + if err := syscall.Kill(-pid, syscall.SIGTERM); err != nil { + _ = syscall.Kill(pid, syscall.SIGTERM) + } + for _, r := range refs { + _ = syscall.Kill(r.pid, syscall.SIGTERM) + } + if left := waitProcRefs(refs, grace); len(left) == 0 { + return nil + } + _ = syscall.Kill(-pid, syscall.SIGKILL) + for _, r := range procRefsAlive(refs) { + _ = syscall.Kill(r.pid, syscall.SIGKILL) + } + return waitProcRefs(refs, settle) +} + +// waitProcRefs attend que tous aient disparu, au plus wait. Renvoie les survivants. +func waitProcRefs(refs []procRef, wait time.Duration) []procRef { + deadline := time.Now().Add(wait) + for { + left := procRefsAlive(refs) + if len(left) == 0 || time.Now().After(deadline) { + return left + } + time.Sleep(100 * time.Millisecond) + } +} diff --git a/internal/loki/sys_proctree_linux_test.go b/internal/loki/sys_proctree_linux_test.go new file mode 100644 index 0000000..aef44c3 --- /dev/null +++ b/internal/loki/sys_proctree_linux_test.go @@ -0,0 +1,83 @@ +//go:build linux + +package loki + +import ( + "os" + "os/exec" + "strconv" + "strings" + "syscall" + "testing" + "time" +) + +func TestParseProcStat(t *testing.T) { + st, ok := parseProcStat("4242 (python3 (x) y) S 4200 4100 4000 0 -1 4194560 0 0 0 0 0 0 0 0 20 0 1 0 987654 0 0") + if !ok || st.state != "S" || st.ppid != 4200 || st.pgrp != 4100 || st.sid != 4000 || st.start != "987654" { + t.Fatalf("lecture fausse : %+v ok=%v", st, ok) + } + if _, ok := parseProcStat("4242 (tronqué"); ok { + t.Error("ligne sans parenthèse fermante acceptée") + } +} + +func TestProcTreeOf(t *testing.T) { + tab := map[int]procStat{ + 100: {ppid: 1, pgrp: 100, sid: 100, start: "a"}, // chef (Setsid) + 101: {ppid: 100, pgrp: 100, sid: 100, start: "b"}, + 200: {ppid: 101, pgrp: 200, sid: 200, start: "c"}, // moteur dans sa propre session + 201: {ppid: 200, pgrp: 200, sid: 200, start: "d"}, + 300: {ppid: 1, pgrp: 300, sid: 300, start: "e"}, // étranger + 400: {ppid: 1, pgrp: 100, sid: 100, start: "f"}, // orphelin rattaché à PID 1 + 500: {ppid: 100, pgrp: 100, sid: 100, start: "g"}, // nous-mêmes + } + got := map[int]bool{} + for _, r := range procTreeOf(tab, 100, 500) { + got[r.pid] = true + } + for _, pid := range []int{100, 101, 200, 201, 400} { + if !got[pid] { + t.Errorf("PID %d oublié", pid) + } + } + if got[300] || got[500] { + t.Errorf("hors de l'arbre pris : %v", got) + } +} + +// Le chef meurt vite, son enfant (dans une autre session, comme le moteur de +// Strata) traîne à s'arrêter : stopProcTree ne revient qu'une fois tout parti. +func TestStopProcTreeWaitsForChildren(t *testing.T) { + // Le chef lance un enfant qui ignore SIGTERM dans sa propre session, écrit + // son PID, puis attend. + pidFile := t.TempDir() + "/child" + script := `setsid sh -c 'trap "" TERM; echo $$ > ` + pidFile + `; while :; do sleep 0.1; done' & wait` + cmd := exec.Command("sh", "-c", script) + cmd.SysProcAttr = &syscall.SysProcAttr{Setsid: true} + if err := cmd.Start(); err != nil { + t.Skip("sh indisponible :", err) + } + go cmd.Wait() // pas de zombie pour le chef + var child int + for i := 0; i < 50 && child == 0; i++ { + time.Sleep(50 * time.Millisecond) + if b, err := os.ReadFile(pidFile); err == nil { + child, _ = strconv.Atoi(strings.TrimSpace(string(b))) + } + } + if child == 0 { + t.Fatal("l'enfant n'a pas démarré") + } + st, ok := readProcStat(child) + if !ok || st.sid == cmd.Process.Pid { + t.Fatalf("l'enfant devait avoir sa propre session : %+v", st) + } + ref := procRef{pid: child, start: st.start} + if left := stopProcTree(cmd.Process.Pid, 300*time.Millisecond, 5*time.Second); len(left) != 0 { + t.Fatalf("survivants : %v", left) + } + if procRefAlive(ref) { + t.Fatal("l'enfant tourne encore au retour de stopProcTree") + } +} diff --git a/internal/loki/sys_service_container.go b/internal/loki/sys_service_container.go index 4c59ac1..6f36d43 100644 --- a/internal/loki/sys_service_container.go +++ b/internal/loki/sys_service_container.go @@ -125,25 +125,25 @@ func userCheckStarted(pid int) error { func userSvcStop(verbose bool) error { pid := readServicePID() - if pid <= 0 || !processAlive(pid) { + // Le chef parti, ses enfants peuvent tourner encore (moteur de Strata) : + // on ne conclut « rien à arrêter » que si l'arbre entier est vide. + if pid <= 0 || len(procTreeOf(procTable(), pid, os.Getpid())) == 0 { _ = os.Remove(pidFilePath()) if verbose { fmt.Println(yellow("[info]") + " aucun service en cours d'exécution") } return nil } - // Setsid a fait de l'enfant un chef de groupe : le PID négatif vise le - // groupe entier, donc llama-server s'arrête avec lui. - if err := syscall.Kill(-pid, syscall.SIGTERM); err != nil { - _ = syscall.Kill(pid, syscall.SIGTERM) - } - for i := 0; i < 40 && processAlive(pid); i++ { - time.Sleep(100 * time.Millisecond) - } - if processAlive(pid) { - _ = syscall.Kill(-pid, syscall.SIGKILL) - } + // Setsid a fait de l'enfant un chef de session : tout ce qu'il a lancé + // (llama-server, serveur et moteur de Strata) s'arrête avec lui, et on + // attend qu'il soit VRAIMENT parti — la relance qui suit en a besoin pour + // trouver la VRAM et la RAM libres. + left := stopProcTree(pid, 4*time.Second, 20*time.Second) _ = os.Remove(pidFilePath()) + if len(left) > 0 { + fmt.Fprintf(os.Stderr, "%s %d processus du moteur toujours là après SIGKILL (PID %d…) : "+ + "la mémoire du GPU peut ne pas être encore rendue\n", yellow("[attention]"), len(left), left[0].pid) + } if verbose { fmt.Println(green("[ok]") + " arrêté") } diff --git a/internal/loki/web_api.go b/internal/loki/web_api.go index 97e9b3f..fafdf6e 100644 --- a/internal/loki/web_api.go +++ b/internal/loki/web_api.go @@ -107,6 +107,13 @@ func modelLoadErrorFrom(log string) string { // On ne considère que ce qui suit la DERNIÈRE tentative de chargement. start := 0 for i, l := range lines { + // Une pile d'appels cite les fonctions traversées + // (…server_context_impl10load_model…) : ce n'est pas une nouvelle + // tentative, et repartir de là perdait la cause écrite juste avant + // (« cudaMalloc failed: out of memory » devenait « a planté »). + if isBacktraceFrame(l) { + continue + } // Un lancement refusé par LOAD_GUARD (backend_serve_moe.go) s'arrête // avant tout chargement : c'est lui, la dernière tentative. if strings.Contains(l, "loading model") || strings.Contains(l, "load_model") || strings.Contains(l, loadGuardMarker) || @@ -159,7 +166,8 @@ func modelLoadErrorFrom(log string) string { strings.Contains(low, "cudamalloc failed"), strings.Contains(low, "failed to allocate"): // Prioritaire sur l'échec générique : la cause mémoire doit gagner. - reason = "mémoire GPU insuffisante (VRAM) — réduis le contexte (CTX) ou prends une quantification plus petite" + reason = "mémoire GPU insuffisante (VRAM) — réduis le contexte (CTX) ou prends une quantification plus petite, " + + "et vérifie qu'aucun autre programme n'occupe la carte (nvidia-smi)" case strings.Contains(low, "ggml_cuda error"), strings.Contains(low, "cuda error"): if reason == "" { reason = "erreur CUDA — GPU indisponible ? (après un redémarrage du serveur : docker restart loki)" @@ -196,6 +204,10 @@ func modelLoadErrorFrom(log string) string { } } +// isBacktraceFrame : une ligne de pile d'appels de glibc +// (« /chemin/lib.so(symbole+0x1c)[0x14bd…] »). +func isBacktraceFrame(l string) bool { return strings.Contains(l, ")[0x") } + // handleServiceLog (GET /api/service/log) : dernières lignes du journal du // service, pour que l'UI puisse montrer POURQUOI le modèle ne se charge pas // (fichier .gguf illisible, VRAM insuffisante, bibliothèque manquante…) au lieu