diff --git a/internal/loki/chat_tplprobe.go b/internal/loki/chat_tplprobe.go new file mode 100644 index 0000000..bee0a9b --- /dev/null +++ b/internal/loki/chat_tplprobe.go @@ -0,0 +1,516 @@ +package loki + +import ( + "bytes" + "context" + "encoding/json" + "errors" + "fmt" + "hash/fnv" + "io" + "maps" + "net/http" + "os" + "path/filepath" + "strings" + "sync" + "time" +) + +// Sonde de gabarit : que fait le gabarit de chat du modèle chargé de ce qu'on +// lui envoie ? Deux questions, celles dont dépendent les pistes de réutilisation +// du cache : +// +// - preservesHistory : le raisonnement d'un message assistant PASSÉ (avant le +// dernier message utilisateur) est-il encore rendu ? Les gabarits Qwen3 le +// retirent, d'autres le gardent. +// - prefixStable : quand un message utilisateur s'ajoute après un tour +// d'outil, le rendu précédent reste-t-il un préfixe exact du nouveau ? Sinon +// le moteur recalcule depuis l'endroit où les deux divergent, quoi que fasse +// son cache. +// +// PUREMENT DIAGNOSTIQUE. Rien ici ne touche à la construction d'un prompt, et +// Loki ne renvoie aujourd'hui JAMAIS de reasoning_content au modèle (Message n'a +// pas ce champ) : preservesHistory décrit le gabarit, pas ce que Loki fait. Une +// évolution qui s'appuierait sur ces réponses pour ajouter, retirer ou +// réordonner quoi que ce soit dans l'historique devra passer sa propre revue de +// fidélité, et traiter « unknown » comme « garder le comportement d'aujourd'hui ». +// +// Règles de conduite, toutes là pour qu'une sonde ne coûte jamais rien au vrai +// travail : +// +// - moteur LOCAL seulement : un preset externe n'a pas de /apply-template, et +// sonder une API tierce n'apprendrait rien de son gabarit ; +// - POST /apply-template et rien d'autre. Ce rendu est sans état (ni file de +// tâches ni slot), donc il ne peut pas évincer le cache de la conversation. +// Jamais de repli sur une complétion : avec un seul slot, elle effacerait +// précisément le KV qu'on cherche à préserver ; +// - lancée APRÈS une complétion terminée, en tâche de fond, une fois /health à +// 200 : jamais sur le chemin du premier jeton ; +// - les mêmes outils et chat_template_kwargs que la complétion qui l'a +// déclenchée : les gabarits Qwen bifurquent sur enable_thinking, sur +// reasoning_effort et sur la présence d'outils ; +// - add_generation_prompt=false demandé au moteur, plutôt que de deviner à la +// main où commence l'amorce de réponse ; +// - tout échec (404 d'un moteur ancien ou d'un fork, 401, exception du +// gabarit, délai) donne « unknown », jamais un oui ou un non. + +type tplTri string + +const ( + tplYes tplTri = "yes" + tplNo tplTri = "no" + tplUnknown tplTri = "unknown" +) + +// tplProbeResult : ce que la sonde a conclu pour un moteur, un modèle, un +// gabarit et une forme de requête donnés. Aucun texte de conversation : les +// messages rendus sont synthétiques, et seuls des verdicts sont gardés. +type tplProbeResult struct { + PreservesHistory tplTri `json:"preserves_history"` + PrefixStable tplTri `json:"prefix_stable"` + Build string `json:"build,omitempty"` + Model string `json:"model,omitempty"` // nom du fichier, pas le chemin + TemplateHash string `json:"template_hash,omitempty"` + Tools int `json:"tools"` + // Kwargs : chat_template_kwargs avec lesquels les rendus ont été faits. + Kwargs map[string]any `json:"chat_template_kwargs,omitempty"` + Effort string `json:"reasoning_effort,omitempty"` + Note string `json:"note,omitempty"` + At time.Time `json:"at"` + // cacheable : la réponse ne changera pas tant que le moteur, le modèle et le + // gabarit restent les mêmes (rendu obtenu, 404, exception du gabarit). Un + // délai, un 503 ou un 401 se retentent plus tard. + cacheable bool +} + +func tplUnknownResult(note string) tplProbeResult { + return tplProbeResult{PreservesHistory: tplUnknown, PrefixStable: tplUnknown, Note: note, At: time.Now()} +} + +// tplShape : la forme de la requête à reproduire. Les outils sont figés en JSON +// dès la prise de forme : la goroutine de sonde ne partage rien avec runChat. +type tplShape struct { + tools json.RawMessage // nil = aucun outil envoyé + nTools int + first string // nom du premier outil, pour un appel vraisemblable + kwargs map[string]any + effort string +} + +func newTplShape(tools []Tool, kwargs map[string]any, effort string) tplShape { + s := tplShape{nTools: len(tools), effort: effort} + if len(tools) > 0 { + s.tools, _ = json.Marshal(tools) + s.first = tools[0].Function.Name + } + if kwargs != nil { + s.kwargs = maps.Clone(kwargs) + } + return s +} + +func (s tplShape) hash() string { + h := fnv.New64a() + h.Write(s.tools) + kw, _ := json.Marshal(s.kwargs) // clés triées : stable + h.Write([]byte{0}) + h.Write(kw) + h.Write([]byte{0}) + h.Write([]byte(s.effort)) + return fmt.Sprintf("%016x", h.Sum64()) +} + +// Messages synthétiques. Des marqueurs qu'aucun gabarit ne produit de lui-même, +// cherchés tels quels dans le rendu. +const ( + tplMarkSys = "loki-sonde-systeme" + tplMarkA = "loki-sonde-question-a" + tplMarkR1 = "loki-sonde-raisonnement-r1" + tplMarkC1 = "loki-sonde-reponse-c1" + tplMarkT1 = "loki-sonde-resultat-t1" + tplMarkB = "loki-sonde-question-b" + // Identifiant d'appel : 9 caractères alphanumériques, la forme qu'exigent les + // gabarits stricts (Mistral) sous peine de raise_exception. + tplCallID = "a1B2c3D4e" +) + +type tplMsg struct { + Role string `json:"role"` + Content string `json:"content"` + ReasoningContent string `json:"reasoning_content,omitempty"` + ToolCalls []ToolCall `json:"tool_calls,omitempty"` + ToolCallID string `json:"tool_call_id,omitempty"` +} + +const ( + tplProbeTimeout = 3 * time.Second // par requête + tplProbeEvery = time.Minute // au plus une vérification par minute + tplProbeCacheMax = 16 + tplProbe503Tries = 3 +) + +var ( + // tplProbeAuto : déclenchement automatique après les complétions. Coupé par + // les tests du paquet (TestMain), dont les faux moteurs ne répondent pas + // tous en JSON : la sonde y est appelée explicitement. + tplProbeAuto = true + // Attente entre deux essais sur un 503 (modèle en chargement). + tplProbeBackoff = time.Second + + tplProbeMu sync.Mutex + tplProbeCache = map[string]tplProbeResult{} + tplProbeLast *tplProbeResult + tplProbeRunning bool + tplProbeTried time.Time + tplProbeLogged string +) + +// tplProbeKick : appelé par runChat après une complétion COMPLÈTE du fil +// principal, sur le moteur local, avec ce qu'elle a envoyé. Ne bloque jamais : +// au plus une sonde en vol, au plus une vérification par minute (la plupart se +// résument à un GET /props qui retrouve la réponse en cache). +func tplProbeKick(tools []Tool, kwargs map[string]any, effort string) { + if !tplProbeAuto { + return + } + tplProbeMu.Lock() + if tplProbeRunning || time.Since(tplProbeTried) < tplProbeEvery { + tplProbeMu.Unlock() + return + } + tplProbeRunning, tplProbeTried = true, time.Now() + tplProbeMu.Unlock() + shape := newTplShape(tools, kwargs, effort) + go func() { + defer func() { + tplProbeMu.Lock() + tplProbeRunning = false + tplProbeMu.Unlock() + }() + ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second) + defer cancel() + tplProbeEnsure(ctx, shape) + }() +} + +// tplCapsCurrent : le dernier verdict connu, pour l'affichage. Rien pour un +// preset externe : un verdict du moteur local n'y dirait rien de vrai. +func tplCapsCurrent() (tplProbeResult, bool) { + if externalActive() { + return tplProbeResult{}, false + } + tplProbeMu.Lock() + defer tplProbeMu.Unlock() + if tplProbeLast == nil { + return tplProbeResult{}, false + } + return *tplProbeLast, true +} + +// tplProbeEnsure : verdict pour la forme donnée, depuis le cache ou une sonde. +func tplProbeEnsure(ctx context.Context, shape tplShape) tplProbeResult { + if resolveChatEndpoint().External { + return tplUnknownResult("preset externe : pas de sonde") + } + p := tplProber{ + base: fmt.Sprintf("http://localhost:%d", LLMPort()), + auth: authHeader, + client: &http.Client{Timeout: tplProbeTimeout}, + } + if !p.healthy(ctx) { + // Moteur absent ou en chargement : rien à conclure, rien à ranger. + return tplUnknownResult("moteur pas prêt (/health)") + } + props, err := p.props(ctx) + if err != nil { + r := tplUnknownResult("/props illisible : " + err.Error()) + tplProbeStore("", r) + return r + } + h := fnv.New64a() + h.Write([]byte(props.ChatTemplate)) + tmplHash := fmt.Sprintf("%016x", h.Sum64()) + // Clé : le binaire, le modèle ET le gabarit effectivement chargé — un + // --chat-template d'EXTRA_ARGS change le dernier sans toucher aux deux + // autres. Plus la forme de la requête, dont le rendu dépend. + key := props.BuildInfo + "\x00" + props.ModelPath + "\x00" + tmplHash + "\x00" + shape.hash() + tplProbeMu.Lock() + if r, ok := tplProbeCache[key]; ok { + tplProbeLast = &r + tplProbeMu.Unlock() + return r + } + tplProbeMu.Unlock() + + r := p.probe(ctx, shape) + r.Build, r.TemplateHash, r.Tools, r.Kwargs, r.Effort = props.BuildInfo, tmplHash, shape.nTools, shape.kwargs, shape.effort + if props.ModelPath != "" { + r.Model = filepath.Base(props.ModelPath) + } + // Une ligne par verdict nouveau : un échec passager qui se répète chaque + // minute à l'identique n'inonde pas le journal. + line := tplProbeLine(r) + tplProbeMu.Lock() + fresh := line != tplProbeLogged + tplProbeLogged = line + tplProbeMu.Unlock() + if fresh { + fmt.Fprintln(os.Stderr, line) + } + tplProbeStore(key, r) + return r +} + +func tplProbeStore(key string, r tplProbeResult) { + tplProbeMu.Lock() + defer tplProbeMu.Unlock() + tplProbeLast = &r + if key == "" || !r.cacheable { + return + } + if len(tplProbeCache) >= tplProbeCacheMax { + oldest := "" + for k, v := range tplProbeCache { + if oldest == "" || v.At.Before(tplProbeCache[oldest].At) { + oldest = k + } + } + delete(tplProbeCache, oldest) + } + tplProbeCache[key] = r +} + +func tplProbeLine(r tplProbeResult) string { + s := fmt.Sprintf("[tplprobe] build=%s modèle=%s gabarit=%s outils=%d historique_raisonnement=%s préfixe_stable=%s", + orDash(r.Build), orDash(r.Model), orDash(r.TemplateHash), r.Tools, r.PreservesHistory, r.PrefixStable) + if r.Note != "" { + s += " (" + r.Note + ")" + } + return s +} + +func orDash(s string) string { + if s == "" { + return "-" + } + return s +} + +type tplProber struct { + base string + auth func(*http.Request) + client *http.Client +} + +// tplHTTPError : réponse non-200 du moteur. +type tplHTTPError struct { + status int + body string +} + +func (e *tplHTTPError) Error() string { + if e.body != "" { + return fmt.Sprintf("HTTP %d : %s", e.status, e.body) + } + return fmt.Sprintf("HTTP %d", e.status) +} + +// tplDefinitive : l'échec se reproduira à l'identique tant que moteur, modèle +// et gabarit ne changent pas (route absente, gabarit qui lève une exception, +// requête refusée). Un 401/403 dépend de la clé, un 503 du chargement, un +// délai de la charge : ceux-là se retentent. +func tplDefinitive(err error) bool { + var he *tplHTTPError + if !errors.As(err, &he) { + return false + } + switch he.status { + case http.StatusUnauthorized, http.StatusForbidden, http.StatusServiceUnavailable, http.StatusTooManyRequests: + return false + } + return he.status >= 400 && he.status < 600 +} + +func (p tplProber) do(ctx context.Context, method, path string, body []byte) ([]byte, error) { + ctx, cancel := context.WithTimeout(ctx, tplProbeTimeout) + defer cancel() + var rd io.Reader + if body != nil { + rd = bytes.NewReader(body) + } + req, err := http.NewRequestWithContext(ctx, method, p.base+path, rd) + if err != nil { + return nil, err + } + if body != nil { + req.Header.Set("Content-Type", "application/json") + } + // Un moteur protégé par API_KEY répond 401 sans la clé (le piège déjà + // payé par la sonde de vision, chat_screenshot.go). + if p.auth != nil { + p.auth(req) + } + resp, err := p.client.Do(req) + if err != nil { + return nil, err + } + defer resp.Body.Close() + data, err := io.ReadAll(io.LimitReader(resp.Body, 8<<20)) + if err != nil { + return nil, err + } + if resp.StatusCode != http.StatusOK { + msg := strings.TrimSpace(string(data)) + if rs := []rune(msg); len(rs) > 160 { + msg = string(rs[:160]) + "…" + } + return nil, &tplHTTPError{status: resp.StatusCode, body: msg} + } + return data, nil +} + +func (p tplProber) healthy(ctx context.Context) bool { + _, err := p.do(ctx, http.MethodGet, "/health", nil) + return err == nil +} + +type tplProps struct { + BuildInfo string `json:"build_info"` + ModelPath string `json:"model_path"` + ChatTemplate string `json:"chat_template"` +} + +func (p tplProber) props(ctx context.Context) (tplProps, error) { + var out tplProps + data, err := p.do(ctx, http.MethodGet, "/props", nil) + if err != nil { + return out, err + } + if err := json.Unmarshal(data, &out); err != nil { + return out, fmt.Errorf("réponse non JSON") + } + return out, nil +} + +// render : POST /apply-template, relancé sur 503 (modèle en chargement). +func (p tplProber) render(ctx context.Context, shape tplShape, msgs []tplMsg, genPrompt bool) (string, error) { + body := map[string]any{"messages": msgs, "add_generation_prompt": genPrompt} + // Mêmes clés que runChat, et seulement celles qui atteignent le gabarit. + if shape.tools != nil { + body["tools"] = shape.tools + body["parallel_tool_calls"] = false + } + if shape.kwargs != nil { + body["chat_template_kwargs"] = shape.kwargs + } + if shape.effort != "" { + body["reasoning_effort"] = shape.effort + } + raw, err := json.Marshal(body) + if err != nil { + return "", err + } + for try := 1; ; try++ { + data, err := p.do(ctx, http.MethodPost, "/apply-template", raw) + var he *tplHTTPError + if err != nil && errors.As(err, &he) && he.status == http.StatusServiceUnavailable && try < tplProbe503Tries { + select { + case <-ctx.Done(): + return "", ctx.Err() + case <-time.After(time.Duration(try) * tplProbeBackoff): + } + continue + } + if err != nil { + return "", err + } + var out struct { + Prompt *string `json:"prompt"` + } + if json.Unmarshal(data, &out) != nil || out.Prompt == nil { + return "", &tplHTTPError{status: http.StatusOK, body: "réponse sans prompt"} + } + return *out.Prompt, nil + } +} + +// probe : les rendus et leurs verdicts. Chaque verdict ne vaut que si ses +// propres rendus ont abouti ; l'autre peut rester inconnu. +func (p tplProber) probe(ctx context.Context, shape tplShape) tplProbeResult { + r := tplUnknownResult("") + definitive := true + var notes []string + fail := func(what string, err error) { + if !tplDefinitive(err) { + definitive = false + } + notes = append(notes, what+" : "+err.Error()) + } + + sys := tplMsg{Role: "system", Content: tplMarkSys} + userA := tplMsg{Role: "user", Content: tplMarkA} + userB := tplMsg{Role: "user", Content: tplMarkB} + + // 1) Raisonnement d'un tour passé : [système, A, assistant{R1, C1}, B]. + hist, err := p.render(ctx, shape, []tplMsg{sys, userA, + {Role: "assistant", Content: tplMarkC1, ReasoningContent: tplMarkR1}, userB}, false) + switch { + case err != nil: + fail("historique", err) + case !strings.Contains(hist, tplMarkC1) || !strings.Contains(hist, tplMarkB): + notes = append(notes, "historique : rendu incohérent") + case strings.Contains(hist, tplMarkR1): + r.PreservesHistory = tplYes + default: + r.PreservesHistory = tplNo + } + + // 2) Stabilité du préfixe à travers un tour d'outil bien formé : un appel + // avec identifiant, puis son résultat, puis un nouveau message utilisateur. + name := shape.first + if name == "" { + name = "lookup" + } + call := tplMsg{Role: "assistant", Content: tplMarkC1, ReasoningContent: tplMarkR1, + ToolCalls: []ToolCall{{ID: tplCallID, Type: "function", Function: ToolCallFunc{Name: name, Arguments: "{}"}}}} + result := tplMsg{Role: "tool", Content: tplMarkT1, ToolCallID: tplCallID} + before := []tplMsg{sys, userA, call, result} + after := append(append([]tplMsg{}, before...), userB) + + r1, err1 := p.render(ctx, shape, before, false) + var r1g, r2 string + var err2, err3 error + if err1 == nil { + r1g, err2 = p.render(ctx, shape, before, true) + } + if err1 == nil && err2 == nil { + r2, err3 = p.render(ctx, shape, after, false) + } + switch { + case err1 != nil: + fail("préfixe", err1) + case err2 != nil: + fail("préfixe", err2) + case err3 != nil: + fail("préfixe", err3) + case r1 == r1g: + // Un moteur qui ignore add_generation_prompt rend l'amorce de réponse + // dans les deux cas : la comparaison conclurait à tort « instable ». + notes = append(notes, "add_generation_prompt ignoré par le moteur") + case !strings.Contains(r1, tplMarkT1) || !strings.Contains(r2, tplMarkB): + notes = append(notes, "préfixe : rendu incohérent") + case strings.HasPrefix(r2, r1): + r.PrefixStable = tplYes + default: + r.PrefixStable = tplNo + } + if err1 == nil && !strings.Contains(r1, tplMarkR1) && r.PreservesHistory == tplNo { + // Même au dernier tour le raisonnement n'apparaît pas : c'est le moteur + // (ou le gabarit) qui ignore reasoning_content, pas un tri par position. + notes = append(notes, "reasoning_content jamais rendu") + } + r.Note = strings.Join(notes, " ; ") + r.cacheable = definitive + return r +} diff --git a/internal/loki/chat_tplprobe_test.go b/internal/loki/chat_tplprobe_test.go new file mode 100644 index 0000000..e2a584b --- /dev/null +++ b/internal/loki/chat_tplprobe_test.go @@ -0,0 +1,432 @@ +package loki + +import ( + "context" + "encoding/json" + "fmt" + "net/http" + "net/http/httptest" + "net/url" + "os" + "regexp" + "strings" + "sync" + "testing" + "time" +) + +// Les faux moteurs des autres tests ne répondent pas tous en JSON sur /props : +// la sonde automatique (lancée après chaque complétion du fil principal) y est +// coupée, et appelée explicitement ici. +func TestMain(m *testing.M) { + tplProbeAuto = false + os.Exit(m.Run()) +} + +func freshTplProbe(t *testing.T) { + t.Helper() + reset := func() { + tplProbeMu.Lock() + tplProbeCache = map[string]tplProbeResult{} + tplProbeLast = nil + tplProbeRunning = false + tplProbeTried = time.Time{} + tplProbeLogged = "" + tplProbeMu.Unlock() + } + reset() + old := tplProbeBackoff + tplProbeBackoff = time.Millisecond + t.Cleanup(func() { reset(); tplProbeBackoff = old }) +} + +// fakeTpl : un llama-server réduit à /health, /props et /apply-template, avec +// un gabarit façon ChatML écrit à la main. +// +// qwen3 : le raisonnement n'est rendu qu'après le dernier message utilisateur ; +// keep : il est toujours rendu ; +// noagp : comme keep, mais add_generation_prompt est ignoré (vieux moteur). +type fakeTpl struct { + mu sync.Mutex + mode string + key string // Bearer exigé hors /health + applyStatus []int // statuts imposés aux prochains /apply-template (0 = rendu) + applyBody string + paths []string + applies []map[string]any +} + +var reCallID = regexp.MustCompile(`^[A-Za-z0-9]{9}$`) + +func (f *fakeTpl) render(body map[string]any) (string, int) { + msgs, _ := body["messages"].([]any) + lastUser := -1 + for i, m := range msgs { + if m.(map[string]any)["role"] == "user" { + lastUser = i + } + } + var b strings.Builder + if tools, ok := body["tools"].([]any); ok { + fmt.Fprintf(&b, "\n", len(tools)) + } + for i, raw := range msgs { + m := raw.(map[string]any) + role, _ := m["role"].(string) + b.WriteString("<|im_start|>" + role + "\n") + if r, _ := m["reasoning_content"].(string); r != "" && role == "assistant" && (f.mode != "qwen3" || i > lastUser) { + b.WriteString("" + r + "") + } + c, _ := m["content"].(string) + b.WriteString(c) + calls, _ := m["tool_calls"].([]any) + for _, rc := range calls { + call := rc.(map[string]any) + id, _ := call["id"].(string) + // Gabarit strict : identifiant de 9 alphanumériques, résultat qui suit. + if !reCallID.MatchString(id) || i+1 >= len(msgs) || msgs[i+1].(map[string]any)["tool_call_id"] != id { + return `{"error":{"code":500,"message":"Jinja Exception: tool call id invalide"}}`, 500 + } + fn := call["function"].(map[string]any) + b.WriteString("" + fn["name"].(string) + "") + } + b.WriteString("<|im_end|>\n") + } + if agp, _ := body["add_generation_prompt"].(bool); agp || f.mode == "noagp" { + b.WriteString("<|im_start|>assistant\n") + } + out, _ := json.Marshal(map[string]string{"prompt": b.String()}) + return string(out), 200 +} + +func (f *fakeTpl) handler(w http.ResponseWriter, r *http.Request) { + f.mu.Lock() + defer f.mu.Unlock() + f.paths = append(f.paths, r.URL.Path) + if r.URL.Path == "/health" { + _, _ = w.Write([]byte(`{"status":"ok"}`)) + return + } + if f.key != "" && r.Header.Get("Authorization") != "Bearer "+f.key { + w.WriteHeader(http.StatusUnauthorized) + _, _ = w.Write([]byte(`{"error":{"code":401,"message":"Invalid API Key"}}`)) + return + } + switch r.URL.Path { + case "/props": + fmt.Fprintf(w, `{"build_info":"b7000-abc","model_path":"/models/Qwen3-8B-Q8_0.gguf","chat_template":%q}`, f.mode) + case "/apply-template": + var body map[string]any + _ = json.NewDecoder(r.Body).Decode(&body) + f.applies = append(f.applies, body) + if len(f.applyStatus) > 0 { + s := f.applyStatus[0] + f.applyStatus = f.applyStatus[1:] + if s != 0 { + w.WriteHeader(s) + _, _ = w.Write([]byte(f.applyBody)) + return + } + } + out, status := f.render(body) + w.WriteHeader(status) + _, _ = w.Write([]byte(out)) + default: + w.WriteHeader(http.StatusNotFound) + } +} + +func (f *fakeTpl) start(t *testing.T) { + t.Helper() + srv := httptest.NewServer(http.HandlerFunc(f.handler)) + t.Cleanup(srv.Close) + u, _ := url.Parse(srv.URL) + if err := SetConfigKey("PORT", u.Port()); err != nil { + t.Fatal(err) + } +} + +func (f *fakeTpl) count(path string) int { + f.mu.Lock() + defer f.mu.Unlock() + n := 0 + for _, p := range f.paths { + if p == path { + n++ + } + } + return n +} + +// Jamais de complétion : avec un seul slot, elle évincerait le cache de la +// conversation que la sonde est censée servir. +func (f *fakeTpl) assertNoCompletion(t *testing.T) { + t.Helper() + f.mu.Lock() + defer f.mu.Unlock() + for _, p := range f.paths { + if strings.Contains(p, "completion") { + t.Fatalf("la sonde a appelé %s", p) + } + } +} + +func probeTools() []Tool { + return []Tool{{Type: "function", Function: ToolFunction{Name: "bash", Description: "d", Parameters: map[string]any{"type": "object"}}}} +} + +func TestSondeGabaritQwen3(t *testing.T) { + testHome(t) + freshTplProbe(t) + f := &fakeTpl{mode: "qwen3"} + f.start(t) + kw := map[string]any{"enable_thinking": false} + r := tplProbeEnsure(context.Background(), newTplShape(probeTools(), kw, "")) + if r.PreservesHistory != tplNo || r.PrefixStable != tplNo { + t.Fatalf("Qwen3 retire le raisonnement passé : %+v", r) + } + if r.Build != "b7000-abc" || r.Model != "Qwen3-8B-Q8_0.gguf" || r.Tools != 1 || !r.cacheable { + t.Fatalf("métadonnées : %+v", r) + } + f.assertNoCompletion(t) + // Mêmes outils et chat_template_kwargs que la complétion ; jamais + // d'amorce de réponse, sauf le rendu qui vérifie que le moteur l'honore. + gen := 0 + for _, a := range f.applies { + if tools, _ := a["tools"].([]any); len(tools) != 1 { + t.Fatalf("outils non transmis : %v", a["tools"]) + } + if k, _ := a["chat_template_kwargs"].(map[string]any); k["enable_thinking"] != false { + t.Fatalf("chat_template_kwargs : %v", a["chat_template_kwargs"]) + } + if a["add_generation_prompt"] == true { + gen++ + } + } + if len(f.applies) != 4 || gen != 1 { + t.Fatalf("%d rendus dont %d avec amorce", len(f.applies), gen) + } + // Même moteur, même modèle, même gabarit, même forme : réponse en cache. + tplProbeEnsure(context.Background(), newTplShape(probeTools(), kw, "")) + if n := f.count("/apply-template"); n != 4 { + t.Fatalf("cache ignoré : %d rendus", n) + } + // Autre forme (raisonnement non coupé) : nouvelle sonde. + tplProbeEnsure(context.Background(), newTplShape(probeTools(), nil, "")) + if n := f.count("/apply-template"); n != 8 { + t.Fatalf("forme différente non resondée : %d rendus", n) + } + // Exposé dans /api/perf/summary. + rec := httptest.NewRecorder() + handlePerfSummary(rec, httptest.NewRequest(http.MethodGet, "/api/perf/summary", nil)) + var sum struct { + Template *struct { + PreservesHistory string `json:"preserves_history"` + PrefixStable string `json:"prefix_stable"` + } `json:"template"` + } + if err := json.Unmarshal(rec.Body.Bytes(), &sum); err != nil || sum.Template == nil || sum.Template.PrefixStable != "no" { + t.Fatalf("résumé : %s", rec.Body.String()) + } + if strings.Contains(rec.Body.String(), "loki-sonde") { + t.Fatal("un rendu a fuité dans le résumé") + } +} + +func TestSondeGabaritConserveLHistorique(t *testing.T) { + testHome(t) + freshTplProbe(t) + f := &fakeTpl{mode: "keep"} + f.start(t) + r := tplProbeEnsure(context.Background(), newTplShape(nil, nil, "")) + if r.PreservesHistory != tplYes || r.PrefixStable != tplYes { + t.Fatalf("gabarit qui garde tout : %+v", r) + } + for _, a := range f.applies { + if _, ok := a["tools"]; ok { + t.Fatal("outils envoyés alors que la complétion n'en avait pas") + } + } +} + +// Un moteur qui ignore add_generation_prompt ne doit pas faire conclure +// « instable » : l'amorce rendue dans les deux cas fausserait la comparaison. +func TestSondeGabaritAmorceIgnoree(t *testing.T) { + testHome(t) + freshTplProbe(t) + f := &fakeTpl{mode: "noagp"} + f.start(t) + r := tplProbeEnsure(context.Background(), newTplShape(nil, nil, "")) + if r.PrefixStable != tplUnknown || r.PreservesHistory != tplYes { + t.Fatalf("amorce ignorée : %+v", r) + } +} + +// Moteur ancien ou fork sans la route : inconnu, rangé (ça ne changera pas +// avant un autre binaire), et aucun repli sur une complétion. +func TestSondeGabarit404(t *testing.T) { + testHome(t) + freshTplProbe(t) + f := &fakeTpl{mode: "qwen3", applyStatus: []int{404, 404, 404, 404}, applyBody: "Not Found"} + f.start(t) + r := tplProbeEnsure(context.Background(), newTplShape(probeTools(), nil, "")) + if r.PreservesHistory != tplUnknown || r.PrefixStable != tplUnknown || !r.cacheable { + t.Fatalf("404 : %+v", r) + } + f.assertNoCompletion(t) + tplProbeEnsure(context.Background(), newTplShape(probeTools(), nil, "")) + if n := f.count("/apply-template"); n != 2 { + t.Fatalf("404 resondé : %d rendus", n) + } +} + +// Gabarit qui lève une exception (raise_exception) : inconnu, jamais oui/non. +func TestSondeGabaritExceptionDuGabarit(t *testing.T) { + testHome(t) + freshTplProbe(t) + f := &fakeTpl{mode: "qwen3", applyStatus: []int{500, 500, 500, 500}, + applyBody: `{"error":{"code":500,"message":"Jinja Exception: Conversation roles must alternate"}}`} + f.start(t) + r := tplProbeEnsure(context.Background(), newTplShape(probeTools(), nil, "")) + if r.PreservesHistory != tplUnknown || r.PrefixStable != tplUnknown || !strings.Contains(r.Note, "500") { + t.Fatalf("exception du gabarit : %+v", r) + } + f.assertNoCompletion(t) +} + +// Moteur protégé par API_KEY : la sonde s'authentifie comme les autres appels +// internes. Sans clé, inconnu — et pas rangé : la clé peut arriver. +func TestSondeGabarit401(t *testing.T) { + testHome(t) + freshTplProbe(t) + f := &fakeTpl{mode: "keep", key: "sk-loki-test"} + f.start(t) + r := tplProbeEnsure(context.Background(), newTplShape(nil, nil, "")) + if r.PreservesHistory != tplUnknown || r.PrefixStable != tplUnknown || !strings.Contains(r.Note, "401") { + t.Fatalf("401 : %+v", r) + } + if err := writeAPIKey("sk-loki-test"); err != nil { + t.Fatal(err) + } + r = tplProbeEnsure(context.Background(), newTplShape(nil, nil, "")) + if r.PreservesHistory != tplYes || r.PrefixStable != tplYes { + t.Fatalf("avec la clé : %+v", r) + } +} + +// Modèle en chargement : les 503 sont relancés, sans rien conclure entre-temps. +func TestSondeGabarit503Relance(t *testing.T) { + testHome(t) + freshTplProbe(t) + f := &fakeTpl{mode: "qwen3", applyStatus: []int{503, 503}, applyBody: `{"error":{"code":503,"message":"Loading model"}}`} + f.start(t) + r := tplProbeEnsure(context.Background(), newTplShape(nil, nil, "")) + if r.PreservesHistory != tplNo || r.PrefixStable != tplNo { + t.Fatalf("après 503 : %+v", r) + } + // 503 persistant : inconnu, non rangé. + freshTplProbe(t) + f.mu.Lock() + f.applyStatus = []int{503, 503, 503, 503, 503, 503, 503, 503, 503, 503, 503, 503} + f.mu.Unlock() + r = tplProbeEnsure(context.Background(), newTplShape(nil, nil, "")) + if r.PreservesHistory != tplUnknown || r.cacheable { + t.Fatalf("503 persistant : %+v", r) + } +} + +// Preset externe : aucune requête, nulle part, et rien d'affiché. +func TestSondeGabaritPresetExterne(t *testing.T) { + testHome(t) + freshTplProbe(t) + f := &fakeTpl{mode: "keep"} + srv := httptest.NewServer(http.HandlerFunc(f.handler)) + t.Cleanup(srv.Close) + if err := WriteConfig(parseEnv(externalPresetContent(srv.URL, "gpt-4o-mini", "", "", false))); err != nil { + t.Fatal(err) + } + u, _ := url.Parse(srv.URL) + if err := SetConfigKey("PORT", u.Port()); err != nil { + t.Fatal(err) + } + r := tplProbeEnsure(context.Background(), newTplShape(nil, nil, "")) + if r.PreservesHistory != tplUnknown || r.PrefixStable != tplUnknown { + t.Fatalf("preset externe : %+v", r) + } + if n := len(f.paths); n != 0 { + t.Fatalf("preset externe sondé : %v", f.paths) + } + if _, ok := tplCapsCurrent(); ok { + t.Fatal("verdict affiché pour un preset externe") + } +} + +// Bout en bout : une complétion du fil principal déclenche la sonde en tâche +// de fond, avec les outils et kwargs de CETTE requête, sans rien ajouter à la +// requête de complétion. +func TestSondeGabaritDeclencheeParRunChat(t *testing.T) { + testHome(t) + freshTplProbe(t) + freshPerf(t) + tplProbeAuto = true + t.Cleanup(func() { tplProbeAuto = false }) + if err := SetConfigKey("REASONING", "off"); err != nil { + t.Fatal(err) + } + f := &fakeTpl{mode: "qwen3"} + var mu sync.Mutex + var chat map[string]any + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + if r.URL.Path == "/v1/chat/completions" { + mu.Lock() + _ = json.NewDecoder(r.Body).Decode(&chat) + mu.Unlock() + w.Header().Set("Content-Type", "text/event-stream") + _, _ = w.Write([]byte(sseChunk("ok") + + `data: {"choices":[{"delta":{},"finish_reason":"stop"}],"usage":{"prompt_tokens":50,"completion_tokens":1},"timings":{"prompt_n":50,"cache_n":0}}` + "\n\n" + + "data: [DONE]\n\n")) + return + } + f.handler(w, r) + })) + t.Cleanup(srv.Close) + u, _ := url.Parse(srv.URL) + if err := SetConfigKey("PORT", u.Port()); err != nil { + t.Fatal(err) + } + tools := probeTools() + if _, err := runChatTools(context.Background(), []Message{{Role: "user", Content: "salut"}}, tools, 0.7, Caps{}, func(StreamEvent) bool { return true }); err != nil { + t.Fatal(err) + } + deadline := time.Now().Add(5 * time.Second) + var r tplProbeResult + for { + var ok bool + if r, ok = tplCapsCurrent(); ok || time.Now().After(deadline) { + break + } + time.Sleep(10 * time.Millisecond) + } + if r.PrefixStable != tplNo || r.Tools != 1 || r.Kwargs["enable_thinking"] != false { + t.Fatalf("sonde après complétion : %+v", r) + } + f.mu.Lock() + for _, a := range f.applies { + mu.Lock() + same := fmt.Sprint(a["tools"]) == fmt.Sprint(chat["tools"]) && fmt.Sprint(a["chat_template_kwargs"]) == fmt.Sprint(chat["chat_template_kwargs"]) + mu.Unlock() + if !same { + f.mu.Unlock() + t.Fatalf("forme différente de la complétion :\nsonde %v / %v\nchat %v / %v", a["tools"], a["chat_template_kwargs"], chat["tools"], chat["chat_template_kwargs"]) + } + } + f.mu.Unlock() + // La sonde ne laisse aucune trace dans la requête de complétion. + mu.Lock() + defer mu.Unlock() + for k := range chat { + if strings.Contains(k, "template") && k != "chat_template_kwargs" || k == "add_generation_prompt" { + t.Fatalf("clé %q ajoutée à la complétion", k) + } + } +} diff --git a/internal/loki/llm_client.go b/internal/loki/llm_client.go index 3e44a91..ce36ebd 100644 --- a/internal/loki/llm_client.go +++ b/internal/loki/llm_client.go @@ -1869,6 +1869,15 @@ func runChatTools(ctx context.Context, messages []Message, tools []Tool, tempera s.LostAlert = perfAlert(rec, perfLostAlertAt(chatCfg)) } cb(StreamEvent{Stats: &s}) + // Sonde de gabarit (chat_tplprobe.go), en tâche de fond et après coup : + // la complétion est déjà lue, la sonde ne retarde aucun jeton. Fil + // principal seulement, sur une requête de forme normale (pas une + // relance outils coupés ou neutralisés), pour reproduire exactement + // les outils et chat_template_kwargs d'un vrai tour. Rien n'en revient + // dans la requête suivante. + if !ep.External && ptag.kind == perfMain && !disableTools && !toolChoiceNone { + tplProbeKick(tools, reasoningKwargs, reasoningEffort) + } if !ep.External { peakGen = max(peakGen, stats.GenTokens) if lastEst > 0 && stats.PromptTokensTotal > 0 { diff --git a/internal/loki/perf_log.go b/internal/loki/perf_log.go index 5d1ba83..420632d 100644 --- a/internal/loki/perf_log.go +++ b/internal/loki/perf_log.go @@ -472,6 +472,9 @@ type perfSummary struct { DraftN int `json:"draft_n"` DraftAccepted int `json:"draft_accepted"` DraftRate *float64 `json:"draft_rate,omitempty"` + // Template : dernier verdict de la sonde de gabarit (chat_tplprobe.go), + // absent tant qu'aucune n'a tourné ou sur un preset externe. + Template *tplProbeResult `json:"template,omitempty"` } var perfDepthBuckets = []struct { @@ -580,5 +583,9 @@ func handlePerfSummary(w http.ResponseWriter, r *http.Request) { sendJSON(w, http.StatusMethodNotAllowed, map[string]any{"error": "GET seulement"}) return } - sendJSON(w, http.StatusOK, perfSummarize(perfLog.snapshot())) + s := perfSummarize(perfLog.snapshot()) + if r, ok := tplCapsCurrent(); ok { + s.Template = &r + } + sendJSON(w, http.StatusOK, s) }