diff --git a/internal/loki/backend_serve.go b/internal/loki/backend_serve.go index ceb74f1..c7b924b 100644 --- a/internal/loki/backend_serve.go +++ b/internal/loki/backend_serve.go @@ -344,11 +344,20 @@ func cmdServe(args []string) error { probeFidelityEnv(&si) probeCacheRAM(cfg, extra, &si) probeCkptEnv(&si) - // Build du moteur : seulement pour un hybride, le seul cas où il décide de - // quelque chose (voir ckptArgs) — inutile de lancer « --version » sinon. - if ggufHybrid(si.GGUF) { + spec := specMode(cfg) + if spec == "auto" || spec == "mtp" { + probeSpec(cfg, &si) + } + // Build du moteur : seulement pour un hybride ou SPEC=auto, les seuls cas où + // il décide de quelque chose (voir ckptArgs, specAutoBlocker) — inutile de + // lancer « --version » sinon. + if ggufHybrid(si.GGUF) || spec == "auto" { si.EngineBuild = engineBuildTrusted(bin) } + specMark := specAutoMark{FP: configFingerprint(cfg), Bin: bin, Build: si.EngineBuild} + if spec == "auto" { + si.SpecAutoBlocked = specAutoCheck(specMark) + } if strings.Contains(si.Help, "--slot-save-path") && !hasAnyFlag(extra, "--slot-save-path") { si.SlotDir = prepareSlotDir(LokiHome()) } @@ -374,6 +383,11 @@ func cmdServe(args []string) error { // lui que le moteur accélère ou non. kt, vt, _ := effectiveKVTypes(cfg, extra, si.ArgEnv) warnSlowKV(kt, vt) + // Jeton de tentative : posé au dernier moment, port libre, l'ancien moteur + // est donc bien parti — le process web ne peut pas le confondre avec lui. + if _, _, auto := specArgs(cfg, extra, si); auto { + specAutoAttempt(specMark) + } fmt.Fprintf(os.Stderr, "[loki serve] %s model=%s port=%s\n", bin, filepath.Base(model), port) @@ -409,8 +423,16 @@ type serveSysInfo struct { SlotDir string // dossier de --slot-save-path, créé et vérifié ; vide = pas de drapeau // Build d'un moteur officiel (engineBuildTrusted), lu seulement pour un - // modèle hybride ; 0 = inconnu ou non lu : ni avis ni drapeau automatique. + // modèle hybride ou SPEC=auto ; 0 = inconnu ou non lu : ni avis ni drapeau + // automatique. EngineBuild int + + // Décodage spéculatif (voir backend_serve_spec.go), lu seulement si SPEC le + // demande. + Draft string // MODEL_DRAFT résolu et vérifié ; vide = absent ou introuvable + DraftErr string // pourquoi MODEL_DRAFT n'a pas été trouvé + DraftGGUF *GGUFInfo // métadonnées du brouillon ; nil = illisibles + SpecAutoBlocked string // un essai automatique précédent a échoué : la raison } // cudaDeviceEnv : sélection GPU (loki gpu), on filtre les devices visibles par @@ -606,6 +628,11 @@ func buildServeArgs(cfg map[string]string, extra []string, bin string, si serveS ck, ckNotes := ckptArgs(cfg, extra, si) args = append(args, ck...) notes = append(notes, ckNotes...) + // Décodage spéculatif (voir backend_serve_spec.go) : rien tant que SPEC + // n'est pas posé. + sp, spNotes, _ := specArgs(cfg, extra, si) + args = append(args, sp...) + notes = append(notes, spNotes...) // EXTRA_ARGS (déjà découpé comme le ferait le shell — les guillemets gardent // ensemble un chemin qui contient des espaces) ferme la marche. args = append(args, extra...) diff --git a/internal/loki/backend_serve_expert.go b/internal/loki/backend_serve_expert.go index 833bc78..95e4a7f 100644 --- a/internal/loki/backend_serve_expert.go +++ b/internal/loki/backend_serve_expert.go @@ -114,14 +114,7 @@ func fitTargetArgs(cfg map[string]string, extra []string, si serveSysInfo) (args case si.ArgEnv["LLAMA_ARG_FIT_TARGET"] != "": return ignore("LLAMA_ARG_FIT_TARGET déjà posé dans l'environnement") } - ctx := flagValue(extra, "-c", "--ctx-size") - if ctx == "" { - ctx = strings.TrimSpace(cfg["CTX"]) - if ctx == "" { - ctx = "32768" - } - } - if n, err := strconv.Atoi(ctx); err != nil || n <= 0 { + if ctx, ok := ctxFixed(cfg, extra); !ok { return ignore("contexte « " + ctx + " » non fixé — --fit pourrait le réduire pour tenir ; donne un CTX chiffré") } if why := fitBlocker(cfg, extra, si); why != "" { diff --git a/internal/loki/backend_serve_spec.go b/internal/loki/backend_serve_spec.go new file mode 100644 index 0000000..92160c3 --- /dev/null +++ b/internal/loki/backend_serve_spec.go @@ -0,0 +1,482 @@ +package loki + +import ( + "encoding/json" + "fmt" + "os" + "regexp" + "strconv" + "strings" + "sync" + "time" + + bolt "go.etcd.io/bbolt" +) + +// Décodage spéculatif géré par Loki : une tête MTP (prédiction de plusieurs +// jetons) intégrée au modèle ou publiée à part, ou un petit modèle brouillon. +// +// SPEC=off (défaut) → rien, même avec MODEL_DRAFT +// SPEC=auto → seulement si tout est réuni (voir specAutoBlocker) +// SPEC=mtp → imposé, sans les garde-fous d'auto, avec un avertissement +// MODEL_DRAFT= → la tête ou le brouillon, résolu comme MMPROJ +// SPEC_N_MAX= → --spec-draft-n-max (défaut du moteur : 3) +// SPEC_SAMPLING=probabilistic → seulement avec SPEC=mtp ; greedy sinon +// +// Rien n'y touche au modèle : chaque jeton émis est tiré par l'échantillonneur +// du modèle cible, un jeton du brouillon n'est gardé que s'il coïncide +// (common_sampler_sample_and_accept_n). Le prix est ailleurs : 1 à 2 Go de VRAM +// en plus, un prefill parfois plus lent, et une fonction très récente qui peut +// empêcher le moteur de démarrer. D'où l'opt-in, les garde-fous d'auto et le +// jeton de tentative (specAutoVerdict) : un lancement automatique qui n'a +// jamais répondu ne se retente pas avec la même configuration. + +const ( + // specAutoMinBuild : premier build officiel où Loki ose l'auto (MTP serveur, + // PR #22673 et suites). Build inconnu = non : un compilé maison ou un fork + // peut connaître le drapeau sans le tenir. + specAutoMinBuild = 11009 + specNMaxLimit = 64 +) + +// specArgEnv : les variables qui règlent la spéculation sans drapeau. Posées, +// elles valent un réglage d'EXTRA_ARGS : Loki se tait. +var specArgEnv = []string{"LLAMA_ARG_SPEC_TYPE", "LLAMA_ARG_SPEC_DRAFT_MODEL", "LLAMA_ARG_SPEC_DRAFT_HF_REPO", + "LLAMA_ARG_SPEC_DRAFT_N_MAX", "LLAMA_ARG_SPEC_DRAFT_SAMPLING"} + +// specUserFlags : la spéculation réglée à la main. Les ajouter EN PLUS ferait +// cumuler les types (--spec-type s'additionne) ou charger deux brouillons. +var specUserFlags = []string{"--spec-type", "-md", "--model-draft", "--spec-draft-model", + "-hfd", "-hfrd", "--spec-draft-hf", "--hf-repo-draft", "--spec-default"} + +// probeSpec : variables du moteur, puis MODEL_DRAFT résolu comme MMPROJ (nom +// simple cherché dans les dossiers déclarés, ou chemin absolu). Introuvable ne +// bloque PAS le lancement, contrairement au projecteur : le brouillon n'est +// qu'une accélération, le moteur démarre sans et la note dit pourquoi. +func probeSpec(cfg map[string]string, si *serveSysInfo) { + if si.ArgEnv == nil { + si.ArgEnv = map[string]string{} + } + for _, k := range specArgEnv { + if v, ok := os.LookupEnv(k); ok && strings.TrimSpace(v) != "" { + si.ArgEnv[k] = v + } + } + d := strings.TrimSpace(cfg["MODEL_DRAFT"]) + if d == "" { + return + } + p, err := resolveServeModelPath(d) + if err == nil { + _, err = os.Stat(p) + } + if err != nil { + si.DraftErr = err.Error() + return + } + si.Draft = p + if g, err := ggufMeta(p); err == nil { + si.DraftGGUF = &g + } +} + +// specMode lit SPEC : "off", "auto", "mtp", ou "" si illisible. +func specMode(cfg map[string]string) string { + switch v := strings.ToLower(strings.TrimSpace(cfg["SPEC"])); v { + case "", "off", "non", "no", "0": + return "off" + case "auto", "mtp": + return v + } + return "" +} + +// specUserSet : EXTRA_ARGS ou l'environnement règlent déjà la spéculation. +func specUserSet(extra []string, argEnv map[string]string) bool { + if hasAnyFlag(extra, specUserFlags...) { + return true + } + for _, k := range []string{"LLAMA_ARG_SPEC_TYPE", "LLAMA_ARG_SPEC_DRAFT_MODEL", "LLAMA_ARG_SPEC_DRAFT_HF_REPO"} { + if argEnv[k] != "" { + return true + } + } + return false +} + +// helpSupportsMTP : le moteur connaît le type draft-mtp ET le drapeau actuel +// du nombre de jetons — un moteur qui n'a que l'ancien --draft-max est d'une +// autre génération, où un MTP n'existe pas. +func helpSupportsMTP(help string) bool { + return strings.Contains(help, "draft-mtp") && strings.Contains(help, "--spec-draft-n-max") +} + +// ctxFixed : le contexte que verra le moteur (EXTRA_ARGS, sinon CTX, sinon le +// défaut de buildServeArgs), et s'il est chiffré. Avec 0 ou une valeur +// illisible, --fit a le droit de RÉDUIRE le contexte pour tenir. +func ctxFixed(cfg map[string]string, extra []string) (string, bool) { + c := flagValue(extra, "-c", "--ctx-size") + if c == "" { + c = strings.TrimSpace(cfg["CTX"]) + } + if c == "" { + c = "32768" + } + n, err := strconv.Atoi(c) + return c, err == nil && n > 0 +} + +// statefulSampler : un échantillonneur dont l'état avance à chaque tirage +// (mirostat, adaptive-p). Le rejet probabiliste fait d'abord tirer le modèle +// cible, puis peut émettre un autre jeton : ces échantillonneurs avanceraient +// sur un jeton que la réponse ne contient pas. Le mode greedy, lui, reste exact. +func statefulSampler(extra []string) string { + for _, v := range flagValues(extra, "--mirostat") { + if v = strings.TrimSpace(v); v != "" && v != "0" { + return "--mirostat " + v + } + } + for _, v := range flagValues(extra, "--samplers") { + if strings.Contains(strings.ToLower(strings.ReplaceAll(v, "-", "_")), "adaptive_p") { + return "--samplers adaptive-p" + } + } + for _, v := range flagValues(extra, "--sampling-seq", "--sampler-seq") { + if strings.Contains(v, "a") { + return "--sampling-seq avec adaptive-p" + } + } + return "" +} + +// specAutoBlocker dit pourquoi SPEC=auto ne s'active pas, ou vide. Chaque +// garde répond à une panne précise : +// +// - qwen4exp : MTP fusionné il y a quelques jours, OOM signalés en multi-GPU ; +// - placement manuel (couches fixées, -ts, -ot, experts sur CPU, -sm row, +// --fit off, -dev) : seul --fit compte la VRAM de la tête MTP ; +// - contexte non chiffré : fit aurait le droit de le RÉDUIRE pour faire de +// la place — de l'information perdue pour le modèle ; +// - projecteur vision : une VRAM déjà serrée par un second modèle ; +// - build officiel inconnu ou trop ancien ; +// - un essai précédent avec la même configuration n'a jamais répondu. +func specAutoBlocker(cfg map[string]string, extra []string, si serveSysInfo) string { + if si.GGUF != nil && strings.EqualFold(si.GGUF.Arch, "qwen4exp") { + return "qwen4exp : MTP trop récent pour ce moteur, SPEC=mtp pour l'imposer" + } + if why := fitBlocker(cfg, extra, si); why != "" { + return "placement manuel (" + why + ") — --fit ne compte pas la VRAM du brouillon" + } + if hasAnyFlag(extra, "-dev", "--device") || si.ArgEnv["LLAMA_ARG_DEVICE"] != "" { + return "placement manuel (--device) — --fit ne compte pas la VRAM du brouillon" + } + if _, ok := ctxFixed(cfg, extra); !ok { + return "contexte non chiffré — fit pourrait le réduire pour loger le brouillon" + } + if si.MMProj != "" || hasAnyFlag(extra, "--mmproj", "-mm") { + return "projecteur vision chargé" + } + if si.EngineBuild < specAutoMinBuild { + if si.EngineBuild <= 0 { + return "build du moteur inconnu (compilé ou fork)" + } + return fmt.Sprintf("moteur b%d, b%d au moins", si.EngineBuild, specAutoMinBuild) + } + return si.SpecAutoBlocked +} + +// specArgs : drapeaux de décodage spéculatif que Loki ajoute, ce qu'il en dit, +// et s'ils viennent de SPEC=auto (cmdServe pose alors le jeton de tentative). +// Fonction pure ; cmdServe a déjà résolu MODEL_DRAFT et lu les GGUF. +func specArgs(cfg map[string]string, extra []string, si serveSysInfo) (args, notes []string, auto bool) { + mode := specMode(cfg) + draftKey := strings.TrimSpace(cfg["MODEL_DRAFT"]) + switch mode { + case "": + return nil, []string{"SPEC=" + cfg["SPEC"] + " illisible (off, auto ou mtp) : sans décodage spéculatif"}, false + case "off": + if draftKey != "" { + return nil, []string{"MODEL_DRAFT ignoré : SPEC=off (auto ou mtp pour s'en servir)"}, false + } + return nil, nil, false + } + if specUserSet(extra, si.ArgEnv) { + return nil, []string{"décodage spéculatif déjà réglé dans EXTRA_ARGS ou LLAMA_ARG_* : SPEC ignoré"}, false + } + skip := func(why string) ([]string, []string, bool) { + return nil, []string{"SPEC=" + mode + " : " + why + " — sans décodage spéculatif"}, false + } + + // La source : MODEL_DRAFT s'il est posé, sinon la tête MTP du modèle. Dans + // les deux cas, on regarde les TENSEURS, comme llama.cpp : une clé + // nextn_predict_layers seule ne prouve pas que la tête est dans le fichier. + mtp := false + switch { + case draftKey != "": + switch { + case si.Draft == "": + why := "MODEL_DRAFT=" + draftKey + " introuvable" + if si.DraftErr != "" { + why += " (" + si.DraftErr + ")" + } + return skip(why) + case si.DraftGGUF == nil: + return skip("MODEL_DRAFT illisible (GGUF incomplet ou en cours de téléchargement ?)") + case si.DraftGGUF.HasNextNTensor: + if !helpSupportsMTP(si.Help) { + return skip("ce moteur ne connaît pas draft-mtp") + } + // Type explicite : llama.cpp ne le devine que sur la première tranche. + args, mtp = []string{"-md", si.Draft, "--spec-type", "draft-mtp"}, true + default: + if !strings.Contains(si.Help, "--spec-type") || !strings.Contains(si.Help, "draft-simple") { + return skip("ce moteur ne connaît pas --spec-type draft-simple") + } + // Sans type, un brouillon qui n'est pas une tête MTP serait chargé en + // VRAM puis jamais utilisé. + args = []string{"-md", si.Draft, "--spec-type", "draft-simple"} + } + case si.GGUF == nil: + return skip("métadonnées du modèle illisibles") + case !si.GGUF.HasNextNTensor: + if mode == "mtp" || si.GGUF.NextN > 0 { + return skip("pas de tête MTP dans ce fichier (publiée à part ? MODEL_DRAFT=mtp-….gguf)") + } + return nil, nil, false // auto sur un modèle sans MTP : rien à dire + case !helpSupportsMTP(si.Help): + return skip("ce moteur ne connaît pas draft-mtp") + default: + args, mtp = []string{"--spec-type", "draft-mtp"}, true + } + + label := "brouillon " + baseName(si.Draft) + if mtp { + label = "tête MTP" + } + if mode == "auto" { + if why := specAutoBlocker(cfg, extra, si); why != "" { + return skip(why) + } + auto = true + notes = append(notes, "SPEC=auto → "+label+" : sortie inchangée (chaque jeton est vérifié par le modèle), "+ + "~1-2 Go de VRAM en plus. Mesure le prefill ; SPEC=off pour couper.") + } else { + note := "SPEC=mtp → " + label + " imposé : sortie inchangée, ~1-2 Go de VRAM en plus" + if why := fitBlocker(cfg, extra, si); why != "" || hasAnyFlag(extra, "-dev", "--device") { + if why == "" { + why = "--device" + } + note += " ; placement manuel (" + why + ") : --fit ne compte pas le brouillon, " + + "place-le avec -devd / -ngld si la VRAM manque" + } + if _, ok := ctxFixed(cfg, extra); !ok { + note += " ; contexte non chiffré : fit pourrait le réduire pour loger le brouillon" + } + if si.GGUF != nil && strings.EqualFold(si.GGUF.Arch, "qwen4exp") { + note += " ; MTP qwen4exp très récent" + } + notes = append(notes, note) + } + + // Nombre de jetons anticipés : défaut du moteur (3) si la clé est vide. + if v := strings.TrimSpace(cfg["SPEC_N_MAX"]); v != "" { + n, err := strconv.Atoi(v) + switch { + case err != nil || n < 1 || n > specNMaxLimit: + notes = append(notes, "SPEC_N_MAX="+v+" illisible (1 à 64 jetons) : défaut du moteur") + case !strings.Contains(si.Help, "--spec-draft-n-max"): + notes = append(notes, "ce moteur ne connaît pas --spec-draft-n-max : SPEC_N_MAX ignoré") + case hasAnyFlag(extra, "--spec-draft-n-max") || si.ArgEnv["LLAMA_ARG_SPEC_DRAFT_N_MAX"] != "": + default: + args = append(args, "--spec-draft-n-max", strconv.Itoa(n)) + } + } + + // Tirage du brouillon. greedy est exact quelle que soit la chaîne + // d'échantillonnage : on le fixe dès que le moteur connaît le drapeau, pour + // qu'un changement de défaut en amont ne change rien ici. + sampling := "greedy" + switch v := strings.ToLower(strings.TrimSpace(cfg["SPEC_SAMPLING"])); v { + case "", "greedy": + case "probabilistic": + switch { + case mode != "mtp": + notes = append(notes, "SPEC_SAMPLING=probabilistic ignoré avec SPEC=auto (SPEC=mtp pour le choisir) : greedy") + case statefulSampler(extra) != "": + notes = append(notes, "SPEC_SAMPLING=probabilistic refusé avec "+statefulSampler(extra)+ + " (cet échantillonneur avancerait sur des jetons rejetés) : greedy") + case !strings.Contains(si.Help, "probabilistic"): + notes = append(notes, "ce moteur ne connaît pas --spec-draft-sampling probabilistic : greedy") + default: + sampling = v + } + default: + notes = append(notes, "SPEC_SAMPLING="+v+" illisible (greedy ou probabilistic) : greedy") + } + if strings.Contains(si.Help, "--spec-draft-sampling") && !hasAnyFlag(extra, "--spec-draft-sampling") && + si.ArgEnv["LLAMA_ARG_SPEC_DRAFT_SAMPLING"] == "" { + args = append(args, "--spec-draft-sampling", sampling) + } + return args, notes, auto +} + +// --- Jeton de tentative ------------------------------------------------------ +// +// Un essai automatique qui fait tomber le moteur ne doit pas se rejouer : sous +// systemd il redémarrerait en boucle en rechargeant 15 Go à chaque fois. cmdServe +// ne peut pas constater l'échec lui-même (sous Unix il DEVIENT llama-server). +// Il pose donc un jeton avant de lancer ; le process web l'efface dès que le +// moteur répond. Au lancement suivant, un jeton encore là pour la même +// configuration, le même moteur et le même build veut dire « jamais répondu » : +// l'auto est coupé pour cette combinaison, et on le dit. Une mise à jour du +// moteur ou un preset modifié retente. Sans process web (CLI seule), le jeton +// n'est jamais effacé : on retombe sur off, le côté sûr. + +const ( + specAttemptKey = "spec_auto_attempt" + specFailedKey = "spec_auto_failed" + specFailedMax = 32 +) + +// specAutoMark identifie une combinaison preset × moteur. +type specAutoMark struct { + FP string `json:"fp"` + Bin string `json:"bin"` + Build int `json:"build"` +} + +func (m specAutoMark) key() string { return fmt.Sprintf("%s|%s|%d", m.FP, m.Bin, m.Build) } + +// specAutoVerdict : pure. Renvoie la raison de couper l'auto (vide = permis) +// et s'il faut inscrire cur parmi les échecs. +func specAutoVerdict(cur specAutoMark, attempt *specAutoMark, failed map[string]string) (why string, record bool) { + if w := failed[cur.key()]; w != "" { + return w, false + } + if attempt != nil && attempt.key() == cur.key() { + return "le dernier lancement avec cette configuration n'a jamais répondu", true + } + return "", false +} + +// specAutoCheck lit l'état, tranche et range : jeton consommé, échec inscrit. +func specAutoCheck(cur specAutoMark) string { + why := "" + _ = update(bkState, func(b *bolt.Bucket) error { + var attempt *specAutoMark + if raw := b.Get([]byte(specAttemptKey)); raw != nil { + var a specAutoMark + if json.Unmarshal(raw, &a) == nil { + attempt = &a + } + } + failed := map[string]string{} + if raw := b.Get([]byte(specFailedKey)); raw != nil { + _ = json.Unmarshal(raw, &failed) + } + w, record := specAutoVerdict(cur, attempt, failed) + why = w + if attempt != nil { + _ = b.Delete([]byte(specAttemptKey)) + } + if !record { + return nil + } + return putFailed(b, failed, cur.key(), w) + }) + return why +} + +func putFailed(b *bolt.Bucket, failed map[string]string, key, why string) error { + if len(failed) >= specFailedMax { + failed = map[string]string{} + } + failed[key] = why + raw, err := json.Marshal(failed) + if err != nil { + return err + } + return b.Put([]byte(specFailedKey), raw) +} + +// specAutoAttempt pose le jeton juste avant de lancer le moteur. +func specAutoAttempt(cur specAutoMark) { + _ = putJSON(bkState, specAttemptKey, cur) +} + +// --- Côté process web -------------------------------------------------------- + +var ( + specWatchOnce sync.Once + offloadedRe = regexp.MustCompile(`offloaded (\d+)/(\d+) layers to GPU`) +) + +// startSpecAttemptWatch : une goroutine qui efface le jeton dès que le moteur +// répond, sans attendre qu'une page interroge /api/status. +func startSpecAttemptWatch() { + specWatchOnce.Do(func() { + go func() { + t := time.NewTicker(5 * time.Second) + defer t.Stop() + for range t.C { + specAttemptTick() + } + }() + }) +} + +func specAttemptTick() { + var a specAutoMark + if !getJSON(bkState, specAttemptKey, &a) { + return + } + if externalActive() { + _ = putBytes(bkState, specAttemptKey, nil) // plus de moteur local à attendre + return + } + if !healthCheck() { + return + } + // Répondu. Mais avec -c fixé, fit ne réduit pas le contexte : s'il manque + // la place du brouillon, il descend des couches en RAM — le moteur tourne, + // plus lentement que sans MTP. Ça compte aussi comme un échec de l'auto. + why := "" + if n, m := lastOffload(serviceLogTail(600)); m > 0 && n < m { + why = fmt.Sprintf("seulement %d/%d couches sur GPU avec le brouillon : fit a déplacé des couches en RAM", n, m) + fmt.Println("[loki] SPEC=auto : " + why + " — coupé au prochain démarrage (SPEC=mtp pour l'imposer)") + } + _ = update(bkState, func(b *bolt.Bucket) error { + _ = b.Delete([]byte(specAttemptKey)) + if why == "" { + return nil + } + failed := map[string]string{} + if raw := b.Get([]byte(specFailedKey)); raw != nil { + _ = json.Unmarshal(raw, &failed) + } + return putFailed(b, failed, a.key(), why) + }) +} + +// lastOffload lit « offloaded N/M layers to GPU » du DERNIER chargement du +// journal (après la dernière ligne load_model). 0, 0 = introuvable : on ne +// conclut rien. +func lastOffload(log string) (n, m int) { + lines := strings.Split(log, "\n") + start := -1 + for i, l := range lines { + if strings.Contains(l, "loading model") || strings.Contains(l, "load_model") { + start = i + } + } + if start < 0 { + return 0, 0 + } + for _, l := range lines[start:] { + if s := offloadedRe.FindStringSubmatch(l); s != nil { + n, _ = strconv.Atoi(s[1]) + m, _ = strconv.Atoi(s[2]) + } + } + return n, m +} diff --git a/internal/loki/backend_serve_spec_test.go b/internal/loki/backend_serve_spec_test.go new file mode 100644 index 0000000..9b1a457 --- /dev/null +++ b/internal/loki/backend_serve_spec_test.go @@ -0,0 +1,270 @@ +package loki + +import ( + "reflect" + "slices" + "strings" + "testing" +) + +const helpSpec = helpRecent + `--spec-draft-n-max N number of tokens to draft for speculative decoding (default: 3) +--spec-draft-sampling {greedy,probabilistic} + how the draft is sampled (default: greedy) +--spec-type [none,ngram-cache,ngram-simple,ngram-map-k,ngram-map-k4v,ngram-mod,draft-simple,draft-eagle3,draft-mtp,draft-dflash,draft-dspark] + comma-separated list of types of speculative decoding to use (default: none) +` + +// helpSpecOld : un moteur qui connaît --spec-type mais pas encore le MTP. +const helpSpecOld = helpRecent + `--spec-type [none,ngram-cache,ngram-simple,draft-simple] +` + +// mtpReady : un Qwen3.6 27B avec sa tête MTP dans le fichier, placé par --fit, +// sur un moteur officiel récent. Le cas où SPEC=auto a le droit d'agir. +func mtpReady() serveSysInfo { + return serveSysInfo{ + Help: helpSpec, + Model: "/models/Qwen3.6-27B-MTP.gguf", + GGUF: &GGUFInfo{Arch: "qwen35", BlockCount: 65, NextN: 1, HasNextNTensor: true, Hybrid: true}, + ArgEnv: map[string]string{}, + EngineBuild: 11351, + } +} + +func TestSpecArgs(t *testing.T) { + mtp := []string{"--spec-type", "draft-mtp", "--spec-draft-sampling", "greedy"} + auto := map[string]string{"SPEC": "auto"} + forced := map[string]string{"SPEC": "mtp"} + with := func(base map[string]string, kv ...string) map[string]string { + m := map[string]string{} + for k, v := range base { + m[k] = v + } + for i := 0; i+1 < len(kv); i += 2 { + m[kv[i]] = kv[i+1] + } + return m + } + cases := []struct { + name string + cfg map[string]string + extra []string + si func(*serveSysInfo) + want []string + wantAuto bool + notes int + }{ + {name: "SPEC absent : rien, même sur un modèle MTP (opt-in)", cfg: map[string]string{}}, + {name: "SPEC=off : rien", cfg: map[string]string{"SPEC": "off"}}, + {name: "SPEC=off avec MODEL_DRAFT : rien, et on le dit", + cfg: map[string]string{"SPEC": "off", "MODEL_DRAFT": "mtp-Qwen3.6-27B.gguf"}, notes: 1}, + {name: "SPEC illisible : rien, et on le dit", cfg: map[string]string{"SPEC": "turbo"}, notes: 1}, + {name: "auto, tête MTP dans le fichier, --fit actif : draft-mtp", cfg: auto, want: mtp, wantAuto: true, notes: 1}, + {name: "auto, clé nextn sans tenseur (tête publiée à part) : rien", + cfg: auto, notes: 1, si: func(s *serveSysInfo) { s.GGUF.HasNextNTensor = false }}, + {name: "auto, modèle sans MTP : rien, sans bruit", + cfg: auto, si: func(s *serveSysInfo) { s.GGUF = &GGUFInfo{Arch: "llama", BlockCount: 32} }}, + {name: "auto, métadonnées illisibles : rien", cfg: auto, notes: 1, si: func(s *serveSysInfo) { s.GGUF = nil }}, + {name: "--spec-type dans EXTRA_ARGS : Loki se tait", cfg: auto, extra: []string{"--spec-type", "ngram-mod"}, notes: 1}, + {name: "--model-draft=… dans EXTRA_ARGS : Loki se tait", cfg: auto, extra: []string{"--model-draft=d.gguf"}, notes: 1}, + {name: "-hfd dans EXTRA_ARGS : Loki se tait", cfg: auto, extra: []string{"-hfd", "u/r"}, notes: 1}, + {name: "--spec-default dans EXTRA_ARGS : Loki se tait", cfg: auto, extra: []string{"--spec-default"}, notes: 1}, + {name: "LLAMA_ARG_SPEC_TYPE posée : Loki se tait", cfg: auto, notes: 1, + si: func(s *serveSysInfo) { s.ArgEnv["LLAMA_ARG_SPEC_TYPE"] = "draft-mtp" }}, + {name: "auto, --tensor-split : rien", cfg: auto, extra: []string{"--tensor-split", "0.965,0.035"}, notes: 1}, + {name: "auto, -ts=… : rien", cfg: auto, extra: []string{"-ts=1,1"}, notes: 1}, + {name: "auto, NGL chiffré : rien", cfg: with(auto, "NGL", "40"), notes: 1}, + {name: "auto, -ngl 99 dans EXTRA_ARGS : rien", cfg: auto, extra: []string{"-ngl", "99"}, notes: 1}, + {name: "auto, -ot : rien", cfg: auto, extra: []string{"-ot", "exps=CPU"}, notes: 1}, + {name: "auto, --n-cpu-moe : rien", cfg: auto, extra: []string{"--n-cpu-moe", "20"}, notes: 1}, + {name: "auto, -cmoe : rien", cfg: auto, extra: []string{"-cmoe"}, notes: 1}, + {name: "auto, -sm row : rien", cfg: auto, extra: []string{"-sm", "row"}, notes: 1}, + {name: "auto, -fit off : rien", cfg: auto, extra: []string{"-fit", "off"}, notes: 1}, + {name: "auto, -dev : rien", cfg: auto, extra: []string{"-dev", "CUDA0"}, notes: 1}, + {name: "auto, LLAMA_ARG_DEVICE : rien", cfg: auto, notes: 1, + si: func(s *serveSysInfo) { s.ArgEnv["LLAMA_ARG_DEVICE"] = "CUDA0" }}, + {name: "auto, CTX=0 (fit pourrait réduire le contexte) : rien", cfg: with(auto, "CTX", "0"), notes: 1}, + {name: "auto, projecteur vision : rien", cfg: auto, notes: 1, + si: func(s *serveSysInfo) { s.MMProj = "/models/mmproj-F16.gguf" }}, + {name: "auto, build inconnu : rien", cfg: auto, notes: 1, si: func(s *serveSysInfo) { s.EngineBuild = 0 }}, + {name: "auto, moteur trop ancien : rien", cfg: auto, notes: 1, si: func(s *serveSysInfo) { s.EngineBuild = 10900 }}, + {name: "auto, qwen4exp : jamais d'office", cfg: auto, notes: 1, si: func(s *serveSysInfo) { s.GGUF.Arch = "qwen4exp" }}, + {name: "auto, essai précédent sans réponse : rien", cfg: auto, notes: 1, + si: func(s *serveSysInfo) { + s.SpecAutoBlocked = "le dernier lancement avec cette configuration n'a jamais répondu" + }}, + {name: "auto, moteur sans draft-mtp : rien", cfg: auto, notes: 1, si: func(s *serveSysInfo) { s.Help = helpSpecOld }}, + {name: "auto, aide illisible : rien", cfg: auto, notes: 1, si: func(s *serveSysInfo) { s.Help = "" }}, + {name: "mtp imposé malgré -ts : drapeaux, avec l'avertissement", cfg: forced, + extra: []string{"-ts", "0.965,0.035"}, want: mtp, notes: 1}, + {name: "mtp imposé sur qwen4exp, build inconnu", cfg: forced, want: mtp, notes: 1, + si: func(s *serveSysInfo) { s.GGUF.Arch = "qwen4exp"; s.EngineBuild = 0 }}, + {name: "mtp imposé, moteur sans draft-mtp : rien", cfg: forced, notes: 1, si: func(s *serveSysInfo) { s.Help = helpSpecOld }}, + {name: "MODEL_DRAFT tête MTP à part : -md chemin absolu + draft-mtp", + cfg: with(auto, "MODEL_DRAFT", "mtp-Qwen3.6-27B-Q8_0.gguf"), wantAuto: true, notes: 1, + want: []string{"-md", "/models/mtp-Qwen3.6-27B-Q8_0.gguf", "--spec-type", "draft-mtp", "--spec-draft-sampling", "greedy"}, + si: func(s *serveSysInfo) { + s.GGUF.HasNextNTensor = false + s.Draft = "/models/mtp-Qwen3.6-27B-Q8_0.gguf" + s.DraftGGUF = &GGUFInfo{Arch: "qwen35", BlockCount: 65, HasNextNTensor: true} + }}, + {name: "MODEL_DRAFT petit modèle : -md + draft-simple, sinon chargé pour rien", + cfg: with(forced, "MODEL_DRAFT", "Qwen3-0.6B.gguf"), notes: 1, + want: []string{"-md", "/models/Qwen3-0.6B.gguf", "--spec-type", "draft-simple", "--spec-draft-sampling", "greedy"}, + si: func(s *serveSysInfo) { + s.Draft = "/models/Qwen3-0.6B.gguf" + s.DraftGGUF = &GGUFInfo{Arch: "qwen3", BlockCount: 28} + }}, + {name: "MODEL_DRAFT introuvable : moteur lancé sans, pas d'erreur", + cfg: with(auto, "MODEL_DRAFT", "absent.gguf"), notes: 1, + si: func(s *serveSysInfo) { s.DraftErr = "fichier introuvable" }}, + {name: "MODEL_DRAFT illisible : rien", cfg: with(auto, "MODEL_DRAFT", "mtp-x.gguf"), notes: 1, + si: func(s *serveSysInfo) { s.Draft = "/models/mtp-x.gguf" }}, + {name: "SPEC_N_MAX=2", cfg: with(auto, "SPEC_N_MAX", "2"), wantAuto: true, notes: 1, + want: []string{"--spec-type", "draft-mtp", "--spec-draft-n-max", "2", "--spec-draft-sampling", "greedy"}}, + {name: "SPEC_N_MAX illisible : défaut du moteur", cfg: with(auto, "SPEC_N_MAX", "beaucoup"), want: mtp, wantAuto: true, notes: 2}, + {name: "--spec-draft-n-max dans EXTRA_ARGS : SPEC_N_MAX se tait, MTP reste", + cfg: with(auto, "SPEC_N_MAX", "2"), extra: []string{"--spec-draft-n-max", "4"}, want: mtp, wantAuto: true, notes: 1}, + {name: "probabilistic jamais avec auto", + cfg: with(auto, "SPEC_SAMPLING", "probabilistic"), want: mtp, wantAuto: true, notes: 2}, + {name: "probabilistic avec mtp imposé", cfg: with(forced, "SPEC_SAMPLING", "probabilistic"), notes: 1, + want: []string{"--spec-type", "draft-mtp", "--spec-draft-sampling", "probabilistic"}}, + {name: "probabilistic refusé avec mirostat", cfg: with(forced, "SPEC_SAMPLING", "probabilistic"), + extra: []string{"--mirostat", "2"}, want: mtp, notes: 2}, + {name: "probabilistic refusé avec adaptive-p", cfg: with(forced, "SPEC_SAMPLING", "probabilistic"), + extra: []string{"--samplers", "top_k;adaptive_p"}, want: mtp, notes: 2}, + {name: "--spec-draft-sampling dans EXTRA_ARGS : on ne le double pas", cfg: auto, + extra: []string{"--spec-draft-sampling", "greedy"}, want: []string{"--spec-type", "draft-mtp"}, wantAuto: true, notes: 1}, + {name: "moteur sans --spec-draft-sampling : pas de drapeau", cfg: auto, wantAuto: true, notes: 1, + want: []string{"--spec-type", "draft-mtp"}, + si: func(s *serveSysInfo) { + s.Help = strings.Replace(s.Help, "--spec-draft-sampling {greedy,probabilistic}", "", 1) + }}, + } + for _, c := range cases { + t.Run(c.name, func(t *testing.T) { + si := mtpReady() + if c.si != nil { + c.si(&si) + } + got, notes, isAuto := specArgs(c.cfg, c.extra, si) + if !reflect.DeepEqual(got, c.want) { + t.Errorf("args = %q, attendu %q", got, c.want) + } + if isAuto != c.wantAuto { + t.Errorf("auto = %v, attendu %v", isAuto, c.wantAuto) + } + if len(notes) != c.notes { + t.Errorf("%d notes, attendu %d : %q", len(notes), c.notes, notes) + } + if specMode(c.cfg) == "auto" && slices.Contains(got, "probabilistic") { + t.Errorf("SPEC=auto ne doit jamais choisir le tirage probabiliste : %q", got) + } + for _, a := range got { + if a == "--draft" || a == "--draft-max" || a == "--draft-n" || a == "-ctkd" || a == "-ctvd" { + t.Errorf("drapeau interdit %q", a) + } + } + }) + } +} + +// La spéculation passe avant EXTRA_ARGS, qui garde le dernier mot. +func TestBuildServeArgsSpec(t *testing.T) { + si := mtpReady() + args, _, _ := buildServeArgs(map[string]string{"SPEC": "auto"}, []string{"--jinja"}, "/bin/llama-server", si) + i := slices.Index(args, "--spec-type") + if i < 0 || args[i+1] != "draft-mtp" { + t.Fatalf("--spec-type draft-mtp absent : %q", args) + } + if j := slices.Index(args, "--jinja"); j < i { + t.Errorf("EXTRA_ARGS doit fermer la marche : %q", args) + } + args, _, _ = buildServeArgs(map[string]string{}, nil, "/bin/llama-server", si) + if slices.Contains(args, "--spec-type") { + t.Errorf("SPEC absent : aucun drapeau attendu, %q", args) + } +} + +func TestStatefulSampler(t *testing.T) { + cases := map[string][]string{ + "": {"--mirostat", "0"}, + "--mirostat 1": {"--mirostat=1"}, + "--samplers adaptive-p": {"--samplers", "penalties;adaptive-p"}, + "--sampling-seq avec adaptive-p": {"--sampling-seq", "kpa"}, + } + for want, extra := range cases { + if got := statefulSampler(extra); got != want { + t.Errorf("statefulSampler(%q) = %q, attendu %q", extra, got, want) + } + } + if got := statefulSampler([]string{"--samplers", "top_k;top_p;temperature"}); got != "" { + t.Errorf("chaîne sans état : %q", got) + } +} + +func TestSpecAutoVerdict(t *testing.T) { + cur := specAutoMark{FP: "abc", Bin: "/e/llama-server", Build: 11351} + other := specAutoMark{FP: "abc", Bin: "/e/llama-server", Build: 11400} + if why, rec := specAutoVerdict(cur, nil, nil); why != "" || rec { + t.Errorf("aucun essai : permis, got %q %v", why, rec) + } + if why, rec := specAutoVerdict(cur, &cur, map[string]string{}); why == "" || !rec { + t.Errorf("jeton resté pour la même combinaison : coupé et inscrit, got %q %v", why, rec) + } + if why, rec := specAutoVerdict(cur, &other, map[string]string{}); why != "" || rec { + t.Errorf("jeton d'un autre build : permis (mise à jour = nouvel essai), got %q %v", why, rec) + } + if why, rec := specAutoVerdict(cur, nil, map[string]string{cur.key(): "OOM"}); why != "OOM" || rec { + t.Errorf("échec inscrit : coupé, got %q %v", why, rec) + } +} + +// Le cycle complet sur une vraie base : jeton posé, jamais effacé, puis coupé +// pour cette combinaison seulement. +func TestSpecAutoCheckStore(t *testing.T) { + testHome(t) + cur := specAutoMark{FP: "abc", Bin: "/e/llama-server", Build: 11351} + if why := specAutoCheck(cur); why != "" { + t.Fatalf("base vide : permis, got %q", why) + } + specAutoAttempt(cur) + if why := specAutoCheck(cur); why == "" { + t.Fatal("jeton resté : l'auto doit être coupé") + } + if why := specAutoCheck(cur); why == "" { + t.Error("l'échec doit rester inscrit au lancement suivant") + } + updated := cur + updated.Build = 11400 + if why := specAutoCheck(updated); why != "" { + t.Errorf("moteur mis à jour : nouvel essai permis, got %q", why) + } + // Un jeton d'une autre configuration est jeté sans rien conclure. + specAutoAttempt(specAutoMark{FP: "zzz", Bin: "/e/llama-server", Build: 11351}) + if why := specAutoCheck(updated); why != "" { + t.Errorf("jeton étranger : permis, got %q", why) + } + var a specAutoMark + if getJSON(bkState, specAttemptKey, &a) { + t.Error("le jeton étranger doit être consommé") + } +} + +func TestLastOffload(t *testing.T) { + log := strings.Join([]string{ + "srv load_model: loading model '/models/old.gguf'", + "load_tensors: offloaded 40/65 layers to GPU", + "srv load_model: loading model '/models/new.gguf'", + "load_tensors: offloaded 65/65 layers to GPU", + "main: server is listening", + }, "\n") + if n, m := lastOffload(log); n != 65 || m != 65 { + t.Errorf("dernier chargement seulement : got %d/%d", n, m) + } + if n, m := lastOffload("load_tensors: offloaded 10/65 layers to GPU"); n != 0 || m != 0 { + t.Errorf("sans ligne de chargement, on ne conclut rien : got %d/%d", n, m) + } + if n, m := lastOffload("srv load_model: loading model 'x'\nload_tensors: offloaded 50/65 layers to GPU"); n != 50 || m != 65 { + t.Errorf("chargement partiel : got %d/%d", n, m) + } +} diff --git a/internal/loki/sys_service.go b/internal/loki/sys_service.go index 973b022..31a081f 100644 --- a/internal/loki/sys_service.go +++ b/internal/loki/sys_service.go @@ -87,6 +87,11 @@ var configTemplate = []struct{ key, help string }{ "chacun pèse 70 à 200 Mio de RAM hôte : 8 à 16 si le modèle remplit déjà la RAM"}, {"CKPT_MIN_STEP", "espacement minimal en jetons entre deux points de reprise (--checkpoint-min-step), > 0 ; " + "vide = défaut du moteur, ou 2048 d'office sur un hybride avec un moteur officiel antérieur à b10864"}, + {"SPEC", "décodage spéculatif, sortie inchangée : off (défaut) / auto = tête MTP du modèle ou MODEL_DRAFT si --fit place " + + "tout et qu'aucun essai n'a échoué / mtp = imposé ; ~1-2 Go de VRAM en plus"}, + {"MODEL_DRAFT", "tête MTP publiée à part (mtp-*.gguf) ou petit modèle brouillon, nom ou chemin comme MODEL ; utilisé si SPEC≠off"}, + {"SPEC_N_MAX", "jetons anticipés par étape (--spec-draft-n-max) ; vide = défaut du moteur (3)"}, + {"SPEC_SAMPLING", "tirage du brouillon : greedy (défaut, exact) ; probabilistic seulement avec SPEC=mtp"}, {"REASONING", "passthrough du mode raisonnement (on/auto/deepseek)"}, {"REASONING_PRESERVE", "on/off = garder ou non la réflexion des tours passés dans le gabarit (--reasoning-preserve) ; " + "vide = défaut du moteur (on depuis b10763). Loki ne renvoie pas cette réflexion : off rend l'ancien historique " + diff --git a/internal/loki/ui/index.html b/internal/loki/ui/index.html index d15216b..86ccf0e 100644 --- a/internal/loki/ui/index.html +++ b/internal/loki/ui/index.html @@ -3478,16 +3478,19 @@ html[data-files="1"] #files-btn{color:var(--accent)} -
Décodage spéculatifanticipe des jetons pour accélérer
+ +
Raisonnementréflexion étape par étape @@ -5091,7 +5100,7 @@ async function openItem(kind, key){ document.getElementById('m-quant').value = currentQuantInTextarea(); populateSettings(); attachDownload(); // téléchargement encore en cours côté serveur ? - await Promise.all([loadGpuDevices(), populateModelPicker(), populateMmproj(), + await Promise.all([loadGpuDevices(), populateModelPicker(), populateMmproj(), populateDraft(), populateDlDirs(), refreshInstalled()]); if(seq !== openSeq) return; } @@ -5449,7 +5458,7 @@ async function deleteInstalled(m){ } if(!r.ok){ toast('erreur : ' + (r.error||'')); return; } toast('modèle supprimé — ' + fmtSize(r.freed||0) + ' libérés'); - await Promise.all([populateModelPicker(), populateMmproj(), populateDlDirs(), + await Promise.all([populateModelPicker(), populateMmproj(), populateDraft(), populateDlDirs(), populateModelDirs(), refreshInstalled()]); } // Un .gguf donné est-il déjà sur le disque, complet ? Sert avant de proposer un @@ -5823,7 +5832,7 @@ function populateSettings(){ set('s-reppen', cfgReadKey('REPEAT_PENALTY')); set('s-moe', eaGetValued('--n-cpu-moe')); set('s-spec-n', eaGetValued('--spec-draft-n-max')); - setSpecType(eaGetValued('--spec-type')); + setSpecType(eaGetValued('--spec-type') || specFromKey(cfgReadKey('SPEC'))); const chk = (id,v)=>{ const e=document.getElementById(id); if(e) e.checked=v; }; // Raisonnement : trois états dans le fichier, deux positions sur l'interrupteur. // « off » est une interdiction explicite passée au moteur ; la clé ABSENTE, elle, @@ -5972,7 +5981,14 @@ function syncLoadModeSub(mode){ const el = document.getElementById('s-loadmode-sub'); if(el) el.textContent = LOAD_MODE_SUB[mode] ?? mode; } -// --- Décodage spéculatif (--spec-type / --spec-draft-n-max) ---------------- +// --- Décodage spéculatif (SPEC, MODEL_DRAFT, --spec-type, --spec-draft-n-max) +// « auto » et « MTP » passent par la clé SPEC : Loki cherche la tête MTP dans +// le fichier et pose les garde-fous au lancement. Les autres types restent des +// --spec-type d'EXTRA_ARGS, qui ont toujours le dernier mot côté serveur. +function specFromKey(v){ + v = String(v||'').trim().toLowerCase(); + return v === 'auto' ? 'auto' : v === 'mtp' ? 'draft-mtp' : ''; +} // Sélectionne le type courant. --spec-type accepte en réalité une LISTE séparée // par des virgules ; une valeur composée (ou un type sorti après cette version // de loki) ne correspondrait à aucune option et serait silencieusement effacée @@ -5998,17 +6014,55 @@ function syncSpecRow(){ if(!sel || !row) return; const on = !!sel.value; row.style.display = on ? '' : 'none'; + // Le brouillon ne sert qu'à SPEC (auto / MTP) ; un brouillon déjà posé reste + // visible pour qu'on puisse le retirer. + const dr = document.getElementById('s-draft-row'); + if(dr) dr.style.display = (sel.value === 'auto' || sel.value === 'draft-mtp' || cfgReadKey('MODEL_DRAFT')) ? '' : 'none'; return on; } function onSpecType(){ const v = document.getElementById('s-spec').value; - eaSetValued('--spec-type', v); + const viaKey = v === 'auto' || v === 'draft-mtp'; + // « non » écrit SPEC=off : explicite, et identique au défaut du serveur. + cfgWriteKey('SPEC', v === 'auto' ? 'auto' : v === 'draft-mtp' ? 'mtp' : v ? '' : 'off'); + eaSetValued('--spec-type', viaKey ? '' : v); if(!v){ eaSetValued('--spec-draft-n-max', ''); document.getElementById('s-spec-n').value = ''; } syncSpecRow(); } +// Liste des brouillons : tous les .gguf sauf les projecteurs vision. Une tête +// MTP publiée à part (mtp-*.gguf) est ce qu'on y choisira le plus souvent. +async function populateDraft(){ + const sel = document.getElementById('m-draft'); + if(!sel) return; + const list = await jget('/api/models'); + const cur = readEnvKey(document.getElementById('m-content').value, 'MODEL_DRAFT'); + const items = (list||[]).filter(m => !isMmprojName(m.name)); + const hit = matchModel(cur, items); + let html = ''; + let matched = false; + for(const m of items){ + const on = (m === hit) ? ' selected' : ''; + if(on) matched = true; + html += ''; + } + if(cur && !matched){ + html += ''; + } + sel.innerHTML = html; + syncSpecRow(); +} +// Choisir un brouillon, c'est demander la spéculation : on passe en « auto » si +// elle était coupée (les garde-fous de Loki restent en place). +function onPickDraft(){ + const v = document.getElementById('m-draft').value; + cfgWriteKey('MODEL_DRAFT', v); + const sel = document.getElementById('s-spec'); + if(v && sel && !sel.value){ sel.value = 'auto'; onSpecType(); toast('décodage spéculatif : auto'); } + syncSpecRow(); +} // Replie/déplie l'éditeur du fichier .env (config brute) dans le modal preset. function toggleRaw(){ @@ -6140,7 +6194,7 @@ let dlQueue = []; // fichier. Le fichier a pu atterrir hors du dossier loki : l'option porte alors // le chemin complet, pas le simple nom. async function selectInstalled(fname){ - await Promise.all([populateModelPicker(), populateMmproj(), populateDlDirs(), refreshInstalled()]); + await Promise.all([populateModelPicker(), populateMmproj(), populateDraft(), populateDlDirs(), refreshInstalled()]); const pick = (id, cb)=>{ const s = document.getElementById(id); if(!s) return false; @@ -6148,6 +6202,7 @@ async function selectInstalled(fname){ if(!o) return false; s.value = o.value; cb(); return true; }; + if(/^mtp-/i.test(fname)) return pick('m-draft', onPickDraft); return isMmprojName(fname) ? pick('m-mmproj', onPickMmproj) : pick('m-model', onPickModel); } async function startDownload(){ @@ -6440,11 +6495,27 @@ async function hfPickRepo(repo){ } o.appendChild(note); + // Tête MTP publiée à part (mtp-*.gguf) : même règle que le projecteur, elle + // ne vaut que pour SON modèle. Décochée : la spéculation reste un choix. + let dfChk = null; + const df = (r.drafts||[])[0] && hfBestProjector(r.drafts); + if(df){ + const dn = document.createElement('label'); + dn.className = 'hf-mm'; + dfChk = document.createElement('input'); + dfChk.type = 'checkbox'; + const t = document.createElement('span'); + t.textContent = ' installer aussi la tête MTP (décodage plus rapide, ~1-2 Go de VRAM) — '+df.name+' ('+fmtSize(df.size)+')' + + (installed(df.name) ? ' — déjà sur le disque' : ''); + dn.append(dfChk, t); + o.appendChild(dn); + } + const list = document.createElement('div'); list.className = 'hf-list'; for(const m of r.models){ const row = document.createElement('div'); row.className = 'hf-row'; - row.onclick = ()=>hfInstall(m, mm && mmChk && mmChk.checked ? mm : null); + row.onclick = ()=>hfInstall(m, mm && mmChk && mmChk.checked ? mm : null, df && dfChk && dfChk.checked ? df : null); const n = document.createElement('span'); n.className = 'hf-name'; n.textContent = m.quant || m.name; const meta = document.createElement('span'); meta.className = 'hf-meta'; @@ -6491,14 +6562,14 @@ function hfBestProjector(list){ // Installe : le modèle d'abord, le projecteur ensuite (la file s'en charge). Le // champ Vision du preset ne se remplit correctement que dans cet ordre. -async function hfInstall(model, projector){ +async function hfInstall(model, projector, draft){ // Le fichier est là : rien à télécharger, donc rien à confirmer — la sonde le // constate côté serveur et startDownloadURL le sélectionne. const here = await modelIsOnDisk(model.name); if(!here && model.verdict === 'trop' && !await askConfirm( (model.why||'Ce modèle dépasse la mémoire disponible.')+'\n\nLe téléchargement fonctionnera, mais le moteur risque de ne pas le charger.', {title:'Installer quand même ?', okText:'Installer', danger:true})) return; - dlQueue = projector ? [projector.url] : []; + dlQueue = [projector, draft].filter(Boolean).map(x => x.url); document.getElementById('m-hf-url').value = model.url; await startDownloadURL(model.url); } diff --git a/internal/loki/ui/src/index.tmpl.html b/internal/loki/ui/src/index.tmpl.html index 299b3e5..2c45ade 100644 --- a/internal/loki/ui/src/index.tmpl.html +++ b/internal/loki/ui/src/index.tmpl.html @@ -1255,16 +1255,19 @@ document.documentElement.setAttribute('data-side',localStorage.getItem('loki-sid
-
Décodage spéculatifanticipe des jetons pour accélérer
+ +
Raisonnementréflexion étape par étape diff --git a/internal/loki/ui/src/js/07-models.js b/internal/loki/ui/src/js/07-models.js index 2d6d7f7..73f34ac 100644 --- a/internal/loki/ui/src/js/07-models.js +++ b/internal/loki/ui/src/js/07-models.js @@ -165,7 +165,7 @@ async function openItem(kind, key){ document.getElementById('m-quant').value = currentQuantInTextarea(); populateSettings(); attachDownload(); // téléchargement encore en cours côté serveur ? - await Promise.all([loadGpuDevices(), populateModelPicker(), populateMmproj(), + await Promise.all([loadGpuDevices(), populateModelPicker(), populateMmproj(), populateDraft(), populateDlDirs(), refreshInstalled()]); if(seq !== openSeq) return; } @@ -523,7 +523,7 @@ async function deleteInstalled(m){ } if(!r.ok){ toast('erreur : ' + (r.error||'')); return; } toast('modèle supprimé — ' + fmtSize(r.freed||0) + ' libérés'); - await Promise.all([populateModelPicker(), populateMmproj(), populateDlDirs(), + await Promise.all([populateModelPicker(), populateMmproj(), populateDraft(), populateDlDirs(), populateModelDirs(), refreshInstalled()]); } // Un .gguf donné est-il déjà sur le disque, complet ? Sert avant de proposer un @@ -897,7 +897,7 @@ function populateSettings(){ set('s-reppen', cfgReadKey('REPEAT_PENALTY')); set('s-moe', eaGetValued('--n-cpu-moe')); set('s-spec-n', eaGetValued('--spec-draft-n-max')); - setSpecType(eaGetValued('--spec-type')); + setSpecType(eaGetValued('--spec-type') || specFromKey(cfgReadKey('SPEC'))); const chk = (id,v)=>{ const e=document.getElementById(id); if(e) e.checked=v; }; // Raisonnement : trois états dans le fichier, deux positions sur l'interrupteur. // « off » est une interdiction explicite passée au moteur ; la clé ABSENTE, elle, @@ -1046,7 +1046,14 @@ function syncLoadModeSub(mode){ const el = document.getElementById('s-loadmode-sub'); if(el) el.textContent = LOAD_MODE_SUB[mode] ?? mode; } -// --- Décodage spéculatif (--spec-type / --spec-draft-n-max) ---------------- +// --- Décodage spéculatif (SPEC, MODEL_DRAFT, --spec-type, --spec-draft-n-max) +// « auto » et « MTP » passent par la clé SPEC : Loki cherche la tête MTP dans +// le fichier et pose les garde-fous au lancement. Les autres types restent des +// --spec-type d'EXTRA_ARGS, qui ont toujours le dernier mot côté serveur. +function specFromKey(v){ + v = String(v||'').trim().toLowerCase(); + return v === 'auto' ? 'auto' : v === 'mtp' ? 'draft-mtp' : ''; +} // Sélectionne le type courant. --spec-type accepte en réalité une LISTE séparée // par des virgules ; une valeur composée (ou un type sorti après cette version // de loki) ne correspondrait à aucune option et serait silencieusement effacée @@ -1072,17 +1079,55 @@ function syncSpecRow(){ if(!sel || !row) return; const on = !!sel.value; row.style.display = on ? '' : 'none'; + // Le brouillon ne sert qu'à SPEC (auto / MTP) ; un brouillon déjà posé reste + // visible pour qu'on puisse le retirer. + const dr = document.getElementById('s-draft-row'); + if(dr) dr.style.display = (sel.value === 'auto' || sel.value === 'draft-mtp' || cfgReadKey('MODEL_DRAFT')) ? '' : 'none'; return on; } function onSpecType(){ const v = document.getElementById('s-spec').value; - eaSetValued('--spec-type', v); + const viaKey = v === 'auto' || v === 'draft-mtp'; + // « non » écrit SPEC=off : explicite, et identique au défaut du serveur. + cfgWriteKey('SPEC', v === 'auto' ? 'auto' : v === 'draft-mtp' ? 'mtp' : v ? '' : 'off'); + eaSetValued('--spec-type', viaKey ? '' : v); if(!v){ eaSetValued('--spec-draft-n-max', ''); document.getElementById('s-spec-n').value = ''; } syncSpecRow(); } +// Liste des brouillons : tous les .gguf sauf les projecteurs vision. Une tête +// MTP publiée à part (mtp-*.gguf) est ce qu'on y choisira le plus souvent. +async function populateDraft(){ + const sel = document.getElementById('m-draft'); + if(!sel) return; + const list = await jget('/api/models'); + const cur = readEnvKey(document.getElementById('m-content').value, 'MODEL_DRAFT'); + const items = (list||[]).filter(m => !isMmprojName(m.name)); + const hit = matchModel(cur, items); + let html = ''; + let matched = false; + for(const m of items){ + const on = (m === hit) ? ' selected' : ''; + if(on) matched = true; + html += ''; + } + if(cur && !matched){ + html += ''; + } + sel.innerHTML = html; + syncSpecRow(); +} +// Choisir un brouillon, c'est demander la spéculation : on passe en « auto » si +// elle était coupée (les garde-fous de Loki restent en place). +function onPickDraft(){ + const v = document.getElementById('m-draft').value; + cfgWriteKey('MODEL_DRAFT', v); + const sel = document.getElementById('s-spec'); + if(v && sel && !sel.value){ sel.value = 'auto'; onSpecType(); toast('décodage spéculatif : auto'); } + syncSpecRow(); +} // Replie/déplie l'éditeur du fichier .env (config brute) dans le modal preset. function toggleRaw(){ @@ -1214,7 +1259,7 @@ let dlQueue = []; // fichier. Le fichier a pu atterrir hors du dossier loki : l'option porte alors // le chemin complet, pas le simple nom. async function selectInstalled(fname){ - await Promise.all([populateModelPicker(), populateMmproj(), populateDlDirs(), refreshInstalled()]); + await Promise.all([populateModelPicker(), populateMmproj(), populateDraft(), populateDlDirs(), refreshInstalled()]); const pick = (id, cb)=>{ const s = document.getElementById(id); if(!s) return false; @@ -1222,6 +1267,7 @@ async function selectInstalled(fname){ if(!o) return false; s.value = o.value; cb(); return true; }; + if(/^mtp-/i.test(fname)) return pick('m-draft', onPickDraft); return isMmprojName(fname) ? pick('m-mmproj', onPickMmproj) : pick('m-model', onPickModel); } async function startDownload(){ @@ -1514,11 +1560,27 @@ async function hfPickRepo(repo){ } o.appendChild(note); + // Tête MTP publiée à part (mtp-*.gguf) : même règle que le projecteur, elle + // ne vaut que pour SON modèle. Décochée : la spéculation reste un choix. + let dfChk = null; + const df = (r.drafts||[])[0] && hfBestProjector(r.drafts); + if(df){ + const dn = document.createElement('label'); + dn.className = 'hf-mm'; + dfChk = document.createElement('input'); + dfChk.type = 'checkbox'; + const t = document.createElement('span'); + t.textContent = ' installer aussi la tête MTP (décodage plus rapide, ~1-2 Go de VRAM) — '+df.name+' ('+fmtSize(df.size)+')' + + (installed(df.name) ? ' — déjà sur le disque' : ''); + dn.append(dfChk, t); + o.appendChild(dn); + } + const list = document.createElement('div'); list.className = 'hf-list'; for(const m of r.models){ const row = document.createElement('div'); row.className = 'hf-row'; - row.onclick = ()=>hfInstall(m, mm && mmChk && mmChk.checked ? mm : null); + row.onclick = ()=>hfInstall(m, mm && mmChk && mmChk.checked ? mm : null, df && dfChk && dfChk.checked ? df : null); const n = document.createElement('span'); n.className = 'hf-name'; n.textContent = m.quant || m.name; const meta = document.createElement('span'); meta.className = 'hf-meta'; @@ -1565,14 +1627,14 @@ function hfBestProjector(list){ // Installe : le modèle d'abord, le projecteur ensuite (la file s'en charge). Le // champ Vision du preset ne se remplit correctement que dans cet ordre. -async function hfInstall(model, projector){ +async function hfInstall(model, projector, draft){ // Le fichier est là : rien à télécharger, donc rien à confirmer — la sonde le // constate côté serveur et startDownloadURL le sélectionne. const here = await modelIsOnDisk(model.name); if(!here && model.verdict === 'trop' && !await askConfirm( (model.why||'Ce modèle dépasse la mémoire disponible.')+'\n\nLe téléchargement fonctionnera, mais le moteur risque de ne pas le charger.', {title:'Installer quand même ?', okText:'Installer', danger:true})) return; - dlQueue = projector ? [projector.url] : []; + dlQueue = [projector, draft].filter(Boolean).map(x => x.url); document.getElementById('m-hf-url').value = model.url; await startDownloadURL(model.url); } diff --git a/internal/loki/web_api.go b/internal/loki/web_api.go index 2582220..8255b64 100644 --- a/internal/loki/web_api.go +++ b/internal/loki/web_api.go @@ -112,6 +112,11 @@ func modelLoadError() string { switch { case strings.Contains(low, "model loaded"), strings.Contains(low, "server is listening"): loaded = true + case strings.Contains(low, "failed to create mtp context"), + strings.Contains(low, "failed to load draft model"), + strings.Contains(low, "failed to initialize speculative decoding context"): + // Avant l'échec générique : c'est le brouillon qui manque, pas le modèle. + reason = "le décodage spéculatif n'a pas pu démarrer (tête MTP ou modèle brouillon) — SPEC=off dans le preset" case strings.Contains(l, "has offset") && strings.Contains(l, "expected"): reason = "format de quantification non reconnu par ce moteur" case strings.Contains(low, "unknown model architecture"), diff --git a/internal/loki/web_server.go b/internal/loki/web_server.go index e8f50e6..bf8f67c 100644 --- a/internal/loki/web_server.go +++ b/internal/loki/web_server.go @@ -131,6 +131,9 @@ func newWebMux() *http.ServeMux { // Planificateur des tâches : une goroutine, un tic par minute, dans le process // qui détient la conversation et le modèle (idempotent, cf. sync.Once). StartTaskScheduler() + // Jeton de tentative du décodage spéculatif automatique : effacé dès que le + // moteur répond, sans dépendre d'une page ouverte (voir backend_serve_spec.go). + startSpecAttemptWatch() mux := http.NewServeMux() // Pages publiques : le HTML et le JS ne contiennent aucun secret. Toute la // donnée et toutes les actions passent par /api/* qui, lui, exige la clé.