From 3ff56e01dbb52a68d7bdadd157400c5a243ce23b Mon Sep 17 00:00:00 2001 From: Michael SCHAL Date: Sun, 4 Oct 2026 00:16:41 +0200 Subject: [PATCH] =?UTF-8?q?Moteur=20:=20corrections=20de=20relecture=20du?= =?UTF-8?q?=20MTP=20opt-in=20=E2=80=94=20le=20contr=C3=B4le=20apr=C3=A8s?= =?UTF-8?q?=20chargement=20voit=20enfin=20le=20journal,=20et=20un=20second?= =?UTF-8?q?=20lancement=20ne=20prend=20plus=20le=20jeton=20du=20premier?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit La relecture de 3a6ad6f a trouvé quatre trous, aucun ne touche la sortie du modèle ; tous font que le garde-fou se trompe de verdict. - Couches en RAM : lastOffload repartait de la dernière ligne contenant « load_model » — or llama-server écrit « load_model: initializing… » APRÈS le chargement, et « loading model tensors » aussi pour le brouillon. Le contrôle ne trouvait jamais rien, ou ne regardait que le brouillon. Repère désormais la ligne « loading model '' » du serveur, et la pire des lignes offloaded compte (modèle ou brouillon). - Jeton de tentative : il était consommé (et l'échec inscrit) AVANT le test du port. Un second « loki serve » refusé pendant que le premier chargeait marquait la configuration comme ratée. Lecture d'abord (specAutoPeek), puis rangement une fois le port libre (specAutoSettle). - Process web : il effaçait le jeton sans le relire sous verrou, donc parfois celui d'un lancement plus récent. specAttemptClear ne touche qu'au sien. - MODEL_DRAFT vers une tête EAGLE-3 ou dFlash : Loki imposait draft-simple, que le moteur ne peut pas charger. La note renvoie maintenant à -md et --spec-type dans EXTRA_ARGS. - Une tête MTP choisie comme MODEL_DRAFT compte comme modèle utilisé (« utilisé par », garde à la suppression). - Éditeur : choisir un brouillon sur un ancien preset dont le MTP est en --spec-type dans EXTRA_ARGS fait passer le preset à SPEC=mtp. Sans ça, le brouillon était ignoré. - Tests : un journal réel (initializing après chargement, brouillon chargé à part), un second lancement qui ne fait que lire, le jeton d'un autre lancement gardé, les têtes eagle3/dflash, et MODEL_DRAFT dans les références. Co-Authored-By: Claude Opus 5.5 --- internal/loki/backend_models_refs.go | 7 +- internal/loki/backend_models_refs_test.go | 8 ++ internal/loki/backend_serve.go | 14 ++- internal/loki/backend_serve_spec.go | 124 +++++++++++++++------- internal/loki/backend_serve_spec_test.go | 82 ++++++++++++-- internal/loki/ui/index.html | 3 + internal/loki/ui/src/js/07-models.js | 3 + 7 files changed, 188 insertions(+), 53 deletions(-) diff --git a/internal/loki/backend_models_refs.go b/internal/loki/backend_models_refs.go index e8f914d..49b4a7b 100644 --- a/internal/loki/backend_models_refs.go +++ b/internal/loki/backend_models_refs.go @@ -46,10 +46,11 @@ func modelRefKey(v string) string { } // modelKeysOf liste les .gguf référencés par une configuration : le modèle, le -// projecteur vision, et un --mmproj posé à la main dans EXTRA_ARGS (les presets -// d'avant la clé MMPROJ l'écrivaient là). +// projecteur vision, le brouillon (MODEL_DRAFT : une tête MTP effacée sous le +// moteur coupe la spéculation sans rien dire), et un --mmproj posé à la main +// dans EXTRA_ARGS (les presets d'avant la clé MMPROJ l'écrivaient là). func modelKeysOf(env map[string]string) []string { - vals := []string{env["MODEL"], env["MMPROJ"]} + vals := []string{env["MODEL"], env["MMPROJ"], env["MODEL_DRAFT"]} args := splitArgs(env["EXTRA_ARGS"]) for i, a := range args { if a == "--mmproj" && i+1 < len(args) { diff --git a/internal/loki/backend_models_refs_test.go b/internal/loki/backend_models_refs_test.go index 818c549..cde3f9c 100644 --- a/internal/loki/backend_models_refs_test.go +++ b/internal/loki/backend_models_refs_test.go @@ -142,6 +142,14 @@ func TestProjecteurCompteCommeReference(t *testing.T) { if len(users) != 1 || users[0] != "Vision" { t.Fatalf("--mmproj d'EXTRA_ARGS non compté : %v", users) } + // La tête MTP publiée à part aussi (MODEL_DRAFT). + writeModel(t, modelsDir(), "mtp-Qwen3.6-27B.gguf", 4) + if _, err := SavePreset("", "MTP", `MODEL="m.gguf"`+"\n"+`SPEC=auto`+"\n"+`MODEL_DRAFT="mtp-Qwen3.6-27B.gguf"`+"\n"); err != nil { + t.Fatal(err) + } + if users, _ := modelUsers("mtp-Qwen3.6-27B.gguf"); len(users) != 1 || users[0] != "MTP" { + t.Fatalf("MODEL_DRAFT= non compté : %v", users) + } } // Une famille de tranches incomplète n'est pas « déjà là » : il reste à diff --git a/internal/loki/backend_serve.go b/internal/loki/backend_serve.go index c7b924b..accd5d6 100644 --- a/internal/loki/backend_serve.go +++ b/internal/loki/backend_serve.go @@ -354,9 +354,11 @@ func cmdServe(args []string) error { if ggufHybrid(si.GGUF) || spec == "auto" { si.EngineBuild = engineBuildTrusted(bin) } + // Lu seulement ici ; rien n'est consommé avant que le port soit libre. specMark := specAutoMark{FP: configFingerprint(cfg), Bin: bin, Build: si.EngineBuild} + specRecord := false if spec == "auto" { - si.SpecAutoBlocked = specAutoCheck(specMark) + si.SpecAutoBlocked, specRecord = specAutoPeek(specMark) } if strings.Contains(si.Help, "--slot-save-path") && !hasAnyFlag(extra, "--slot-save-path") { si.SlotDir = prepareSlotDir(LokiHome()) @@ -383,10 +385,12 @@ func cmdServe(args []string) error { // lui que le moteur accélère ou non. kt, vt, _ := effectiveKVTypes(cfg, extra, si.ArgEnv) warnSlowKV(kt, vt) - // Jeton de tentative : posé au dernier moment, port libre, l'ancien moteur - // est donc bien parti — le process web ne peut pas le confondre avec lui. - if _, _, auto := specArgs(cfg, extra, si); auto { - specAutoAttempt(specMark) + // Jeton de tentative : l'ancien consommé (et l'échec inscrit), le nouveau + // posé, au dernier moment — port libre, l'ancien moteur est donc bien parti + // et le process web ne peut pas le confondre avec lui. + if spec == "auto" { + _, _, auto := specArgs(cfg, extra, si) + specAutoSettle(specMark, si.SpecAutoBlocked, specRecord, auto) } fmt.Fprintf(os.Stderr, "[loki serve] %s model=%s port=%s\n", diff --git a/internal/loki/backend_serve_spec.go b/internal/loki/backend_serve_spec.go index 92160c3..8324774 100644 --- a/internal/loki/backend_serve_spec.go +++ b/internal/loki/backend_serve_spec.go @@ -80,6 +80,17 @@ func probeSpec(cfg map[string]string, si *serveSysInfo) { } } +// specSidecarArch : têtes de spéculation qui ne sont pas des modèles autonomes +// (llama.cpp : draft-eagle3, draft-dflash/dspark). Un -md vers elles avec +// draft-simple ne démarrerait pas : leur type se règle à la main. +func specSidecarArch(arch string) bool { + switch strings.ToLower(strings.TrimSpace(arch)) { + case "eagle3", "dflash": + return true + } + return false +} + // specMode lit SPEC : "off", "auto", "mtp", ou "" si illisible. func specMode(cfg map[string]string) string { switch v := strings.ToLower(strings.TrimSpace(cfg["SPEC"])); v { @@ -228,6 +239,9 @@ func specArgs(cfg map[string]string, extra []string, si serveSysInfo) (args, not } // Type explicite : llama.cpp ne le devine que sur la première tranche. args, mtp = []string{"-md", si.Draft, "--spec-type", "draft-mtp"}, true + case specSidecarArch(si.DraftGGUF.Arch): + return skip("MODEL_DRAFT est une tête " + si.DraftGGUF.Arch + + ", pas un modèle brouillon : règle -md et --spec-type dans EXTRA_ARGS") default: if !strings.Contains(si.Help, "--spec-type") || !strings.Contains(si.Help, "draft-simple") { return skip("ce moteur ne connaît pas --spec-type draft-simple") @@ -359,32 +373,56 @@ func specAutoVerdict(cur specAutoMark, attempt *specAutoMark, failed map[string] return "", false } -// specAutoCheck lit l'état, tranche et range : jeton consommé, échec inscrit. -func specAutoCheck(cur specAutoMark) string { - why := "" +// specAutoPeek lit l'état sans rien changer : la raison de couper l'auto (vide +// = permis) et s'il faudra inscrire l'échec. Rien n'est consommé avant que le +// port soit libre (specAutoSettle) : un second « loki serve », refusé parce que +// le premier charge encore, ne doit pas prendre le jeton de celui-ci pour un +// échec. +func specAutoPeek(cur specAutoMark) (why string, record bool) { + _ = view(bkState, func(b *bolt.Bucket) error { + attempt, failed := readSpecAuto(b) + why, record = specAutoVerdict(cur, attempt, failed) + return nil + }) + return why, record +} + +// specAutoSettle range, port libre, juste avant de lancer le moteur : l'ancien +// jeton est consommé, l'échec inscrit s'il y a lieu (record, why), et un +// nouveau jeton posé si ce lancement ajoute des drapeaux automatiques. +func specAutoSettle(cur specAutoMark, why string, record, attempt bool) { _ = update(bkState, func(b *bolt.Bucket) error { - var attempt *specAutoMark - if raw := b.Get([]byte(specAttemptKey)); raw != nil { - var a specAutoMark - if json.Unmarshal(raw, &a) == nil { - attempt = &a + _, failed := readSpecAuto(b) + _ = b.Delete([]byte(specAttemptKey)) + if record { + if err := putFailed(b, failed, cur.key(), why); err != nil { + return err } } - failed := map[string]string{} - if raw := b.Get([]byte(specFailedKey)); raw != nil { - _ = json.Unmarshal(raw, &failed) - } - w, record := specAutoVerdict(cur, attempt, failed) - why = w - if attempt != nil { - _ = b.Delete([]byte(specAttemptKey)) - } - if !record { + if !attempt { return nil } - return putFailed(b, failed, cur.key(), w) + raw, err := json.Marshal(cur) + if err != nil { + return err + } + return b.Put([]byte(specAttemptKey), raw) }) - return why +} + +// readSpecAuto : le jeton en cours (nil = aucun) et les échecs inscrits. +func readSpecAuto(b *bolt.Bucket) (attempt *specAutoMark, failed map[string]string) { + if raw := b.Get([]byte(specAttemptKey)); raw != nil { + var a specAutoMark + if json.Unmarshal(raw, &a) == nil { + attempt = &a + } + } + failed = map[string]string{} + if raw := b.Get([]byte(specFailedKey)); raw != nil { + _ = json.Unmarshal(raw, &failed) + } + return attempt, failed } func putFailed(b *bolt.Bucket, failed map[string]string, key, why string) error { @@ -399,11 +437,6 @@ func putFailed(b *bolt.Bucket, failed map[string]string, key, why string) error return b.Put([]byte(specFailedKey), raw) } -// specAutoAttempt pose le jeton juste avant de lancer le moteur. -func specAutoAttempt(cur specAutoMark) { - _ = putJSON(bkState, specAttemptKey, cur) -} - // --- Côté process web -------------------------------------------------------- var ( @@ -431,7 +464,7 @@ func specAttemptTick() { return } if externalActive() { - _ = putBytes(bkState, specAttemptKey, nil) // plus de moteur local à attendre + specAttemptClear(a, "") // plus de moteur local à attendre return } if !healthCheck() { @@ -445,27 +478,38 @@ func specAttemptTick() { why = fmt.Sprintf("seulement %d/%d couches sur GPU avec le brouillon : fit a déplacé des couches en RAM", n, m) fmt.Println("[loki] SPEC=auto : " + why + " — coupé au prochain démarrage (SPEC=mtp pour l'imposer)") } + specAttemptClear(a, why) +} + +// specAttemptClear efface le jeton a — et lui seul : relu sous verrou, car un +// « loki serve » a pu entre-temps poser celui d'un AUTRE lancement, qui n'a pas +// encore répondu. why non vide inscrit l'échec de a. +func specAttemptClear(a specAutoMark, why string) { _ = update(bkState, func(b *bolt.Bucket) error { + cur, failed := readSpecAuto(b) + if cur == nil || cur.key() != a.key() { + return nil + } _ = b.Delete([]byte(specAttemptKey)) if why == "" { return nil } - failed := map[string]string{} - if raw := b.Get([]byte(specFailedKey)); raw != nil { - _ = json.Unmarshal(raw, &failed) - } return putFailed(b, failed, a.key(), why) }) } -// lastOffload lit « offloaded N/M layers to GPU » du DERNIER chargement du -// journal (après la dernière ligne load_model). 0, 0 = introuvable : on ne +// lastOffload lit les lignes « offloaded N/M layers to GPU » du DERNIER +// chargement du journal, repéré à la ligne du serveur « loading model +// '' ». Pas à « load_model » (le serveur écrit encore « load_model: +// initializing… » APRÈS le chargement) ni à « loading model tensors » (écrit +// aussi pour le brouillon, qui masquerait le modèle). Un brouillon chargé à +// part a sa propre ligne : la pire des deux compte. 0, 0 = introuvable : on ne // conclut rien. func lastOffload(log string) (n, m int) { lines := strings.Split(log, "\n") start := -1 for i, l := range lines { - if strings.Contains(l, "loading model") || strings.Contains(l, "load_model") { + if strings.Contains(l, "loading model '") { start = i } } @@ -473,9 +517,17 @@ func lastOffload(log string) (n, m int) { return 0, 0 } for _, l := range lines[start:] { - if s := offloadedRe.FindStringSubmatch(l); s != nil { - n, _ = strconv.Atoi(s[1]) - m, _ = strconv.Atoi(s[2]) + s := offloadedRe.FindStringSubmatch(l) + if s == nil { + continue + } + a, _ := strconv.Atoi(s[1]) + b, _ := strconv.Atoi(s[2]) + if m == 0 || a < b { + n, m = a, b + } + if a < b { + break } } return n, m diff --git a/internal/loki/backend_serve_spec_test.go b/internal/loki/backend_serve_spec_test.go index 9b1a457..4ff7326 100644 --- a/internal/loki/backend_serve_spec_test.go +++ b/internal/loki/backend_serve_spec_test.go @@ -114,6 +114,18 @@ func TestSpecArgs(t *testing.T) { s.Draft = "/models/Qwen3-0.6B.gguf" s.DraftGGUF = &GGUFInfo{Arch: "qwen3", BlockCount: 28} }}, + {name: "MODEL_DRAFT tête dFlash : pas de draft-simple, réglage à la main", + cfg: with(forced, "MODEL_DRAFT", "dflash-Qwen3.6.gguf"), notes: 1, + si: func(s *serveSysInfo) { + s.Draft = "/models/dflash-Qwen3.6.gguf" + s.DraftGGUF = &GGUFInfo{Arch: "dflash", BlockCount: 5} + }}, + {name: "MODEL_DRAFT tête EAGLE-3 : idem", + cfg: with(auto, "MODEL_DRAFT", "eagle3.gguf"), notes: 1, + si: func(s *serveSysInfo) { + s.Draft = "/models/eagle3.gguf" + s.DraftGGUF = &GGUFInfo{Arch: "eagle3", BlockCount: 1} + }}, {name: "MODEL_DRAFT introuvable : moteur lancé sans, pas d'erreur", cfg: with(auto, "MODEL_DRAFT", "absent.gguf"), notes: 1, si: func(s *serveSysInfo) { s.DraftErr = "fichier introuvable" }}, @@ -223,28 +235,62 @@ func TestSpecAutoVerdict(t *testing.T) { // pour cette combinaison seulement. func TestSpecAutoCheckStore(t *testing.T) { testHome(t) + // Ce que fait cmdServe : lire, puis — port libre — ranger et poser le jeton. + launch := func(cur specAutoMark, wantAttempt bool) string { + why, record := specAutoPeek(cur) + specAutoSettle(cur, why, record, wantAttempt && why == "") + return why + } cur := specAutoMark{FP: "abc", Bin: "/e/llama-server", Build: 11351} - if why := specAutoCheck(cur); why != "" { + if why := launch(cur, true); why != "" { t.Fatalf("base vide : permis, got %q", why) } - specAutoAttempt(cur) - if why := specAutoCheck(cur); why == "" { + // Second « loki serve » refusé (port pris) : il a lu, sans rien ranger. Le + // jeton du premier, qui charge encore, doit rester intact. + if why, _ := specAutoPeek(cur); why == "" { + t.Fatal("jeton présent : le lecteur doit le voir") + } + var a specAutoMark + if !getJSON(bkState, specAttemptKey, &a) || a != cur { + t.Fatal("specAutoPeek ne doit rien consommer") + } + if why := launch(cur, true); why == "" { t.Fatal("jeton resté : l'auto doit être coupé") } - if why := specAutoCheck(cur); why == "" { + if getJSON(bkState, specAttemptKey, &a) { + t.Error("auto coupé : aucun nouveau jeton") + } + if why := launch(cur, true); why == "" { t.Error("l'échec doit rester inscrit au lancement suivant") } updated := cur updated.Build = 11400 - if why := specAutoCheck(updated); why != "" { + if why := launch(updated, true); why != "" { t.Errorf("moteur mis à jour : nouvel essai permis, got %q", why) } + // Le process web efface SON jeton ; pas celui d'un lancement plus récent. + specAttemptClear(cur, "") + if !getJSON(bkState, specAttemptKey, &a) || a != updated { + t.Fatal("le jeton d'un autre lancement doit rester") + } + specAttemptClear(updated, "") + if getJSON(bkState, specAttemptKey, &a) { + t.Fatal("le moteur a répondu : jeton effacé") + } + if why := launch(updated, true); why != "" { + t.Errorf("lancement réussi : permis, got %q", why) + } + // Répondu mais couches en RAM : échec inscrit pour cette combinaison. + specAttemptClear(updated, "couches en RAM") + if why, _ := specAutoPeek(updated); why != "couches en RAM" { + t.Errorf("échec après chargement : coupé, got %q", why) + } // Un jeton d'une autre configuration est jeté sans rien conclure. - specAutoAttempt(specAutoMark{FP: "zzz", Bin: "/e/llama-server", Build: 11351}) - if why := specAutoCheck(updated); why != "" { - t.Errorf("jeton étranger : permis, got %q", why) + other := specAutoMark{FP: "zzz", Bin: "/e/llama-server", Build: 11351} + specAutoSettle(other, "", false, true) + if why := launch(cur, false); why == "" { + t.Error("échec de cur toujours inscrit") } - var a specAutoMark if getJSON(bkState, specAttemptKey, &a) { t.Error("le jeton étranger doit être consommé") } @@ -267,4 +313,22 @@ func TestLastOffload(t *testing.T) { if n, m := lastOffload("srv load_model: loading model 'x'\nload_tensors: offloaded 50/65 layers to GPU"); n != 50 || m != 65 { t.Errorf("chargement partiel : got %d/%d", n, m) } + // Le journal réel de llama-server : « load_model: initializing » APRÈS le + // chargement, et un brouillon chargé à part qui a ses propres lignes. + srvLog := strings.Join([]string{ + "srv load_model: loading model '/models/Qwen3.6-27B.gguf'", + "load_tensors: loading model tensors, this can take a while... (load_mode = mmap)", + "load_tensors: offloaded 52/65 layers to GPU", + "load_tensors: loading model tensors, this can take a while... (load_mode = mmap)", + "load_tensors: offloaded 2/2 layers to GPU", + "srv load_model: initializing, n_slots = 1, n_ctx_slot = 65536, kv_unified = 'false'", + "main: server is listening on http://127.0.0.1:8080", + }, "\n") + if n, m := lastOffload(srvLog); n != 52 || m != 65 { + t.Errorf("modèle partiel, brouillon complet : le pire compte, got %d/%d", n, m) + } + full := strings.Replace(srvLog, "52/65", "65/65", 1) + if n, m := lastOffload(full); n != m || m == 0 { + t.Errorf("tout sur GPU : got %d/%d", n, m) + } } diff --git a/internal/loki/ui/index.html b/internal/loki/ui/index.html index 86ccf0e..ac1ce16 100644 --- a/internal/loki/ui/index.html +++ b/internal/loki/ui/index.html @@ -6061,6 +6061,9 @@ function onPickDraft(){ cfgWriteKey('MODEL_DRAFT', v); const sel = document.getElementById('s-spec'); if(v && sel && !sel.value){ sel.value = 'auto'; onSpecType(); toast('décodage spéculatif : auto'); } + // Un ancien preset portait le MTP en --spec-type dans EXTRA_ARGS : sans clé + // SPEC, le brouillon serait ignoré au lancement. On le range dans SPEC=mtp. + else if(v && sel && sel.value === 'draft-mtp' && specFromKey(cfgReadKey('SPEC')) !== 'draft-mtp') onSpecType(); syncSpecRow(); } diff --git a/internal/loki/ui/src/js/07-models.js b/internal/loki/ui/src/js/07-models.js index 73f34ac..8798b06 100644 --- a/internal/loki/ui/src/js/07-models.js +++ b/internal/loki/ui/src/js/07-models.js @@ -1126,6 +1126,9 @@ function onPickDraft(){ cfgWriteKey('MODEL_DRAFT', v); const sel = document.getElementById('s-spec'); if(v && sel && !sel.value){ sel.value = 'auto'; onSpecType(); toast('décodage spéculatif : auto'); } + // Un ancien preset portait le MTP en --spec-type dans EXTRA_ARGS : sans clé + // SPEC, le brouillon serait ignoré au lancement. On le range dans SPEC=mtp. + else if(v && sel && sel.value === 'draft-mtp' && specFromKey(cfgReadKey('SPEC')) !== 'draft-mtp') onSpecType(); syncSpecRow(); }