mirror of
https://github.com/R0m1k3/Loki.git
synced 2026-10-11 17:26:57 +02:00
Télémétrie : une sonde du gabarit de chat, diagnostique et sans risque pour le cache
Les pistes de réutilisation du cache dépendent de deux faits que rien ne mesurait : le gabarit du modèle rend-il encore le raisonnement d'un tour passé, et le rendu d'un tour d'outil reste-t-il un préfixe exact quand un message utilisateur s'y ajoute ? La sonde pose ces deux questions au moteur local par POST /apply-template — un rendu sans état, qui ne prend ni slot ni place dans la file — et n'en tire que des verdicts oui/non/inconnu. Elle ne touche à aucune construction de prompt : Loki ne renvoie toujours aucun reasoning_content au modèle. Toute évolution qui voudrait s'en servir passera sa propre revue, et lira « unknown » comme « ne rien changer ». - moteur local seulement ; preset externe : aucune requête, rien d'affiché - jamais de complétion de repli : avec un slot, elle évincerait le cache - en tâche de fond après une complétion complète du fil principal, /health à 200, au plus une vérification par minute ; authHeader ; 3 s par requête - mêmes outils, chat_template_kwargs et reasoning_effort que la complétion - add_generation_prompt=false demandé au moteur ; un moteur qui l'ignore donne « inconnu » plutôt qu'un faux « instable » - tour d'outil bien formé (identifiant de 9 alphanumériques, résultat lié) - 404, 401, exception du gabarit, délai : inconnu ; 503 relancé - cache par build_info + modèle + empreinte du gabarit + forme de requête - verdict dans /api/perf/summary (champ template) et une ligne [tplprobe] au journal par verdict nouveau Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
1 parent
09ee04037b
commit
3af481ff05
4 files changed
+965
-1
No files matched your search
@@ -0,0 +1,516 @@
|
||||
package loki
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"hash/fnv"
|
||||
"io"
|
||||
"maps"
|
||||
"net/http"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
)
|
||||
|
||||
// Sonde de gabarit : que fait le gabarit de chat du modèle chargé de ce qu'on
|
||||
// lui envoie ? Deux questions, celles dont dépendent les pistes de réutilisation
|
||||
// du cache :
|
||||
//
|
||||
// - preservesHistory : le raisonnement d'un message assistant PASSÉ (avant le
|
||||
// dernier message utilisateur) est-il encore rendu ? Les gabarits Qwen3 le
|
||||
// retirent, d'autres le gardent.
|
||||
// - prefixStable : quand un message utilisateur s'ajoute après un tour
|
||||
// d'outil, le rendu précédent reste-t-il un préfixe exact du nouveau ? Sinon
|
||||
// le moteur recalcule depuis l'endroit où les deux divergent, quoi que fasse
|
||||
// son cache.
|
||||
//
|
||||
// PUREMENT DIAGNOSTIQUE. Rien ici ne touche à la construction d'un prompt, et
|
||||
// Loki ne renvoie aujourd'hui JAMAIS de reasoning_content au modèle (Message n'a
|
||||
// pas ce champ) : preservesHistory décrit le gabarit, pas ce que Loki fait. Une
|
||||
// évolution qui s'appuierait sur ces réponses pour ajouter, retirer ou
|
||||
// réordonner quoi que ce soit dans l'historique devra passer sa propre revue de
|
||||
// fidélité, et traiter « unknown » comme « garder le comportement d'aujourd'hui ».
|
||||
//
|
||||
// Règles de conduite, toutes là pour qu'une sonde ne coûte jamais rien au vrai
|
||||
// travail :
|
||||
//
|
||||
// - moteur LOCAL seulement : un preset externe n'a pas de /apply-template, et
|
||||
// sonder une API tierce n'apprendrait rien de son gabarit ;
|
||||
// - POST /apply-template et rien d'autre. Ce rendu est sans état (ni file de
|
||||
// tâches ni slot), donc il ne peut pas évincer le cache de la conversation.
|
||||
// Jamais de repli sur une complétion : avec un seul slot, elle effacerait
|
||||
// précisément le KV qu'on cherche à préserver ;
|
||||
// - lancée APRÈS une complétion terminée, en tâche de fond, une fois /health à
|
||||
// 200 : jamais sur le chemin du premier jeton ;
|
||||
// - les mêmes outils et chat_template_kwargs que la complétion qui l'a
|
||||
// déclenchée : les gabarits Qwen bifurquent sur enable_thinking, sur
|
||||
// reasoning_effort et sur la présence d'outils ;
|
||||
// - add_generation_prompt=false demandé au moteur, plutôt que de deviner à la
|
||||
// main où commence l'amorce de réponse ;
|
||||
// - tout échec (404 d'un moteur ancien ou d'un fork, 401, exception du
|
||||
// gabarit, délai) donne « unknown », jamais un oui ou un non.
|
||||
|
||||
type tplTri string
|
||||
|
||||
const (
|
||||
tplYes tplTri = "yes"
|
||||
tplNo tplTri = "no"
|
||||
tplUnknown tplTri = "unknown"
|
||||
)
|
||||
|
||||
// tplProbeResult : ce que la sonde a conclu pour un moteur, un modèle, un
|
||||
// gabarit et une forme de requête donnés. Aucun texte de conversation : les
|
||||
// messages rendus sont synthétiques, et seuls des verdicts sont gardés.
|
||||
type tplProbeResult struct {
|
||||
PreservesHistory tplTri `json:"preserves_history"`
|
||||
PrefixStable tplTri `json:"prefix_stable"`
|
||||
Build string `json:"build,omitempty"`
|
||||
Model string `json:"model,omitempty"` // nom du fichier, pas le chemin
|
||||
TemplateHash string `json:"template_hash,omitempty"`
|
||||
Tools int `json:"tools"`
|
||||
// Kwargs : chat_template_kwargs avec lesquels les rendus ont été faits.
|
||||
Kwargs map[string]any `json:"chat_template_kwargs,omitempty"`
|
||||
Effort string `json:"reasoning_effort,omitempty"`
|
||||
Note string `json:"note,omitempty"`
|
||||
At time.Time `json:"at"`
|
||||
// cacheable : la réponse ne changera pas tant que le moteur, le modèle et le
|
||||
// gabarit restent les mêmes (rendu obtenu, 404, exception du gabarit). Un
|
||||
// délai, un 503 ou un 401 se retentent plus tard.
|
||||
cacheable bool
|
||||
}
|
||||
|
||||
func tplUnknownResult(note string) tplProbeResult {
|
||||
return tplProbeResult{PreservesHistory: tplUnknown, PrefixStable: tplUnknown, Note: note, At: time.Now()}
|
||||
}
|
||||
|
||||
// tplShape : la forme de la requête à reproduire. Les outils sont figés en JSON
|
||||
// dès la prise de forme : la goroutine de sonde ne partage rien avec runChat.
|
||||
type tplShape struct {
|
||||
tools json.RawMessage // nil = aucun outil envoyé
|
||||
nTools int
|
||||
first string // nom du premier outil, pour un appel vraisemblable
|
||||
kwargs map[string]any
|
||||
effort string
|
||||
}
|
||||
|
||||
func newTplShape(tools []Tool, kwargs map[string]any, effort string) tplShape {
|
||||
s := tplShape{nTools: len(tools), effort: effort}
|
||||
if len(tools) > 0 {
|
||||
s.tools, _ = json.Marshal(tools)
|
||||
s.first = tools[0].Function.Name
|
||||
}
|
||||
if kwargs != nil {
|
||||
s.kwargs = maps.Clone(kwargs)
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
func (s tplShape) hash() string {
|
||||
h := fnv.New64a()
|
||||
h.Write(s.tools)
|
||||
kw, _ := json.Marshal(s.kwargs) // clés triées : stable
|
||||
h.Write([]byte{0})
|
||||
h.Write(kw)
|
||||
h.Write([]byte{0})
|
||||
h.Write([]byte(s.effort))
|
||||
return fmt.Sprintf("%016x", h.Sum64())
|
||||
}
|
||||
|
||||
// Messages synthétiques. Des marqueurs qu'aucun gabarit ne produit de lui-même,
|
||||
// cherchés tels quels dans le rendu.
|
||||
const (
|
||||
tplMarkSys = "loki-sonde-systeme"
|
||||
tplMarkA = "loki-sonde-question-a"
|
||||
tplMarkR1 = "loki-sonde-raisonnement-r1"
|
||||
tplMarkC1 = "loki-sonde-reponse-c1"
|
||||
tplMarkT1 = "loki-sonde-resultat-t1"
|
||||
tplMarkB = "loki-sonde-question-b"
|
||||
// Identifiant d'appel : 9 caractères alphanumériques, la forme qu'exigent les
|
||||
// gabarits stricts (Mistral) sous peine de raise_exception.
|
||||
tplCallID = "a1B2c3D4e"
|
||||
)
|
||||
|
||||
type tplMsg struct {
|
||||
Role string `json:"role"`
|
||||
Content string `json:"content"`
|
||||
ReasoningContent string `json:"reasoning_content,omitempty"`
|
||||
ToolCalls []ToolCall `json:"tool_calls,omitempty"`
|
||||
ToolCallID string `json:"tool_call_id,omitempty"`
|
||||
}
|
||||
|
||||
const (
|
||||
tplProbeTimeout = 3 * time.Second // par requête
|
||||
tplProbeEvery = time.Minute // au plus une vérification par minute
|
||||
tplProbeCacheMax = 16
|
||||
tplProbe503Tries = 3
|
||||
)
|
||||
|
||||
var (
|
||||
// tplProbeAuto : déclenchement automatique après les complétions. Coupé par
|
||||
// les tests du paquet (TestMain), dont les faux moteurs ne répondent pas
|
||||
// tous en JSON : la sonde y est appelée explicitement.
|
||||
tplProbeAuto = true
|
||||
// Attente entre deux essais sur un 503 (modèle en chargement).
|
||||
tplProbeBackoff = time.Second
|
||||
|
||||
tplProbeMu sync.Mutex
|
||||
tplProbeCache = map[string]tplProbeResult{}
|
||||
tplProbeLast *tplProbeResult
|
||||
tplProbeRunning bool
|
||||
tplProbeTried time.Time
|
||||
tplProbeLogged string
|
||||
)
|
||||
|
||||
// tplProbeKick : appelé par runChat après une complétion COMPLÈTE du fil
|
||||
// principal, sur le moteur local, avec ce qu'elle a envoyé. Ne bloque jamais :
|
||||
// au plus une sonde en vol, au plus une vérification par minute (la plupart se
|
||||
// résument à un GET /props qui retrouve la réponse en cache).
|
||||
func tplProbeKick(tools []Tool, kwargs map[string]any, effort string) {
|
||||
if !tplProbeAuto {
|
||||
return
|
||||
}
|
||||
tplProbeMu.Lock()
|
||||
if tplProbeRunning || time.Since(tplProbeTried) < tplProbeEvery {
|
||||
tplProbeMu.Unlock()
|
||||
return
|
||||
}
|
||||
tplProbeRunning, tplProbeTried = true, time.Now()
|
||||
tplProbeMu.Unlock()
|
||||
shape := newTplShape(tools, kwargs, effort)
|
||||
go func() {
|
||||
defer func() {
|
||||
tplProbeMu.Lock()
|
||||
tplProbeRunning = false
|
||||
tplProbeMu.Unlock()
|
||||
}()
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
|
||||
defer cancel()
|
||||
tplProbeEnsure(ctx, shape)
|
||||
}()
|
||||
}
|
||||
|
||||
// tplCapsCurrent : le dernier verdict connu, pour l'affichage. Rien pour un
|
||||
// preset externe : un verdict du moteur local n'y dirait rien de vrai.
|
||||
func tplCapsCurrent() (tplProbeResult, bool) {
|
||||
if externalActive() {
|
||||
return tplProbeResult{}, false
|
||||
}
|
||||
tplProbeMu.Lock()
|
||||
defer tplProbeMu.Unlock()
|
||||
if tplProbeLast == nil {
|
||||
return tplProbeResult{}, false
|
||||
}
|
||||
return *tplProbeLast, true
|
||||
}
|
||||
|
||||
// tplProbeEnsure : verdict pour la forme donnée, depuis le cache ou une sonde.
|
||||
func tplProbeEnsure(ctx context.Context, shape tplShape) tplProbeResult {
|
||||
if resolveChatEndpoint().External {
|
||||
return tplUnknownResult("preset externe : pas de sonde")
|
||||
}
|
||||
p := tplProber{
|
||||
base: fmt.Sprintf("http://localhost:%d", LLMPort()),
|
||||
auth: authHeader,
|
||||
client: &http.Client{Timeout: tplProbeTimeout},
|
||||
}
|
||||
if !p.healthy(ctx) {
|
||||
// Moteur absent ou en chargement : rien à conclure, rien à ranger.
|
||||
return tplUnknownResult("moteur pas prêt (/health)")
|
||||
}
|
||||
props, err := p.props(ctx)
|
||||
if err != nil {
|
||||
r := tplUnknownResult("/props illisible : " + err.Error())
|
||||
tplProbeStore("", r)
|
||||
return r
|
||||
}
|
||||
h := fnv.New64a()
|
||||
h.Write([]byte(props.ChatTemplate))
|
||||
tmplHash := fmt.Sprintf("%016x", h.Sum64())
|
||||
// Clé : le binaire, le modèle ET le gabarit effectivement chargé — un
|
||||
// --chat-template d'EXTRA_ARGS change le dernier sans toucher aux deux
|
||||
// autres. Plus la forme de la requête, dont le rendu dépend.
|
||||
key := props.BuildInfo + "\x00" + props.ModelPath + "\x00" + tmplHash + "\x00" + shape.hash()
|
||||
tplProbeMu.Lock()
|
||||
if r, ok := tplProbeCache[key]; ok {
|
||||
tplProbeLast = &r
|
||||
tplProbeMu.Unlock()
|
||||
return r
|
||||
}
|
||||
tplProbeMu.Unlock()
|
||||
|
||||
r := p.probe(ctx, shape)
|
||||
r.Build, r.TemplateHash, r.Tools, r.Kwargs, r.Effort = props.BuildInfo, tmplHash, shape.nTools, shape.kwargs, shape.effort
|
||||
if props.ModelPath != "" {
|
||||
r.Model = filepath.Base(props.ModelPath)
|
||||
}
|
||||
// Une ligne par verdict nouveau : un échec passager qui se répète chaque
|
||||
// minute à l'identique n'inonde pas le journal.
|
||||
line := tplProbeLine(r)
|
||||
tplProbeMu.Lock()
|
||||
fresh := line != tplProbeLogged
|
||||
tplProbeLogged = line
|
||||
tplProbeMu.Unlock()
|
||||
if fresh {
|
||||
fmt.Fprintln(os.Stderr, line)
|
||||
}
|
||||
tplProbeStore(key, r)
|
||||
return r
|
||||
}
|
||||
|
||||
func tplProbeStore(key string, r tplProbeResult) {
|
||||
tplProbeMu.Lock()
|
||||
defer tplProbeMu.Unlock()
|
||||
tplProbeLast = &r
|
||||
if key == "" || !r.cacheable {
|
||||
return
|
||||
}
|
||||
if len(tplProbeCache) >= tplProbeCacheMax {
|
||||
oldest := ""
|
||||
for k, v := range tplProbeCache {
|
||||
if oldest == "" || v.At.Before(tplProbeCache[oldest].At) {
|
||||
oldest = k
|
||||
}
|
||||
}
|
||||
delete(tplProbeCache, oldest)
|
||||
}
|
||||
tplProbeCache[key] = r
|
||||
}
|
||||
|
||||
func tplProbeLine(r tplProbeResult) string {
|
||||
s := fmt.Sprintf("[tplprobe] build=%s modèle=%s gabarit=%s outils=%d historique_raisonnement=%s préfixe_stable=%s",
|
||||
orDash(r.Build), orDash(r.Model), orDash(r.TemplateHash), r.Tools, r.PreservesHistory, r.PrefixStable)
|
||||
if r.Note != "" {
|
||||
s += " (" + r.Note + ")"
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
func orDash(s string) string {
|
||||
if s == "" {
|
||||
return "-"
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
type tplProber struct {
|
||||
base string
|
||||
auth func(*http.Request)
|
||||
client *http.Client
|
||||
}
|
||||
|
||||
// tplHTTPError : réponse non-200 du moteur.
|
||||
type tplHTTPError struct {
|
||||
status int
|
||||
body string
|
||||
}
|
||||
|
||||
func (e *tplHTTPError) Error() string {
|
||||
if e.body != "" {
|
||||
return fmt.Sprintf("HTTP %d : %s", e.status, e.body)
|
||||
}
|
||||
return fmt.Sprintf("HTTP %d", e.status)
|
||||
}
|
||||
|
||||
// tplDefinitive : l'échec se reproduira à l'identique tant que moteur, modèle
|
||||
// et gabarit ne changent pas (route absente, gabarit qui lève une exception,
|
||||
// requête refusée). Un 401/403 dépend de la clé, un 503 du chargement, un
|
||||
// délai de la charge : ceux-là se retentent.
|
||||
func tplDefinitive(err error) bool {
|
||||
var he *tplHTTPError
|
||||
if !errors.As(err, &he) {
|
||||
return false
|
||||
}
|
||||
switch he.status {
|
||||
case http.StatusUnauthorized, http.StatusForbidden, http.StatusServiceUnavailable, http.StatusTooManyRequests:
|
||||
return false
|
||||
}
|
||||
return he.status >= 400 && he.status < 600
|
||||
}
|
||||
|
||||
func (p tplProber) do(ctx context.Context, method, path string, body []byte) ([]byte, error) {
|
||||
ctx, cancel := context.WithTimeout(ctx, tplProbeTimeout)
|
||||
defer cancel()
|
||||
var rd io.Reader
|
||||
if body != nil {
|
||||
rd = bytes.NewReader(body)
|
||||
}
|
||||
req, err := http.NewRequestWithContext(ctx, method, p.base+path, rd)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if body != nil {
|
||||
req.Header.Set("Content-Type", "application/json")
|
||||
}
|
||||
// Un moteur protégé par API_KEY répond 401 sans la clé (le piège déjà
|
||||
// payé par la sonde de vision, chat_screenshot.go).
|
||||
if p.auth != nil {
|
||||
p.auth(req)
|
||||
}
|
||||
resp, err := p.client.Do(req)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
data, err := io.ReadAll(io.LimitReader(resp.Body, 8<<20))
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
msg := strings.TrimSpace(string(data))
|
||||
if rs := []rune(msg); len(rs) > 160 {
|
||||
msg = string(rs[:160]) + "…"
|
||||
}
|
||||
return nil, &tplHTTPError{status: resp.StatusCode, body: msg}
|
||||
}
|
||||
return data, nil
|
||||
}
|
||||
|
||||
func (p tplProber) healthy(ctx context.Context) bool {
|
||||
_, err := p.do(ctx, http.MethodGet, "/health", nil)
|
||||
return err == nil
|
||||
}
|
||||
|
||||
type tplProps struct {
|
||||
BuildInfo string `json:"build_info"`
|
||||
ModelPath string `json:"model_path"`
|
||||
ChatTemplate string `json:"chat_template"`
|
||||
}
|
||||
|
||||
func (p tplProber) props(ctx context.Context) (tplProps, error) {
|
||||
var out tplProps
|
||||
data, err := p.do(ctx, http.MethodGet, "/props", nil)
|
||||
if err != nil {
|
||||
return out, err
|
||||
}
|
||||
if err := json.Unmarshal(data, &out); err != nil {
|
||||
return out, fmt.Errorf("réponse non JSON")
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// render : POST /apply-template, relancé sur 503 (modèle en chargement).
|
||||
func (p tplProber) render(ctx context.Context, shape tplShape, msgs []tplMsg, genPrompt bool) (string, error) {
|
||||
body := map[string]any{"messages": msgs, "add_generation_prompt": genPrompt}
|
||||
// Mêmes clés que runChat, et seulement celles qui atteignent le gabarit.
|
||||
if shape.tools != nil {
|
||||
body["tools"] = shape.tools
|
||||
body["parallel_tool_calls"] = false
|
||||
}
|
||||
if shape.kwargs != nil {
|
||||
body["chat_template_kwargs"] = shape.kwargs
|
||||
}
|
||||
if shape.effort != "" {
|
||||
body["reasoning_effort"] = shape.effort
|
||||
}
|
||||
raw, err := json.Marshal(body)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
for try := 1; ; try++ {
|
||||
data, err := p.do(ctx, http.MethodPost, "/apply-template", raw)
|
||||
var he *tplHTTPError
|
||||
if err != nil && errors.As(err, &he) && he.status == http.StatusServiceUnavailable && try < tplProbe503Tries {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return "", ctx.Err()
|
||||
case <-time.After(time.Duration(try) * tplProbeBackoff):
|
||||
}
|
||||
continue
|
||||
}
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
var out struct {
|
||||
Prompt *string `json:"prompt"`
|
||||
}
|
||||
if json.Unmarshal(data, &out) != nil || out.Prompt == nil {
|
||||
return "", &tplHTTPError{status: http.StatusOK, body: "réponse sans prompt"}
|
||||
}
|
||||
return *out.Prompt, nil
|
||||
}
|
||||
}
|
||||
|
||||
// probe : les rendus et leurs verdicts. Chaque verdict ne vaut que si ses
|
||||
// propres rendus ont abouti ; l'autre peut rester inconnu.
|
||||
func (p tplProber) probe(ctx context.Context, shape tplShape) tplProbeResult {
|
||||
r := tplUnknownResult("")
|
||||
definitive := true
|
||||
var notes []string
|
||||
fail := func(what string, err error) {
|
||||
if !tplDefinitive(err) {
|
||||
definitive = false
|
||||
}
|
||||
notes = append(notes, what+" : "+err.Error())
|
||||
}
|
||||
|
||||
sys := tplMsg{Role: "system", Content: tplMarkSys}
|
||||
userA := tplMsg{Role: "user", Content: tplMarkA}
|
||||
userB := tplMsg{Role: "user", Content: tplMarkB}
|
||||
|
||||
// 1) Raisonnement d'un tour passé : [système, A, assistant{R1, C1}, B].
|
||||
hist, err := p.render(ctx, shape, []tplMsg{sys, userA,
|
||||
{Role: "assistant", Content: tplMarkC1, ReasoningContent: tplMarkR1}, userB}, false)
|
||||
switch {
|
||||
case err != nil:
|
||||
fail("historique", err)
|
||||
case !strings.Contains(hist, tplMarkC1) || !strings.Contains(hist, tplMarkB):
|
||||
notes = append(notes, "historique : rendu incohérent")
|
||||
case strings.Contains(hist, tplMarkR1):
|
||||
r.PreservesHistory = tplYes
|
||||
default:
|
||||
r.PreservesHistory = tplNo
|
||||
}
|
||||
|
||||
// 2) Stabilité du préfixe à travers un tour d'outil bien formé : un appel
|
||||
// avec identifiant, puis son résultat, puis un nouveau message utilisateur.
|
||||
name := shape.first
|
||||
if name == "" {
|
||||
name = "lookup"
|
||||
}
|
||||
call := tplMsg{Role: "assistant", Content: tplMarkC1, ReasoningContent: tplMarkR1,
|
||||
ToolCalls: []ToolCall{{ID: tplCallID, Type: "function", Function: ToolCallFunc{Name: name, Arguments: "{}"}}}}
|
||||
result := tplMsg{Role: "tool", Content: tplMarkT1, ToolCallID: tplCallID}
|
||||
before := []tplMsg{sys, userA, call, result}
|
||||
after := append(append([]tplMsg{}, before...), userB)
|
||||
|
||||
r1, err1 := p.render(ctx, shape, before, false)
|
||||
var r1g, r2 string
|
||||
var err2, err3 error
|
||||
if err1 == nil {
|
||||
r1g, err2 = p.render(ctx, shape, before, true)
|
||||
}
|
||||
if err1 == nil && err2 == nil {
|
||||
r2, err3 = p.render(ctx, shape, after, false)
|
||||
}
|
||||
switch {
|
||||
case err1 != nil:
|
||||
fail("préfixe", err1)
|
||||
case err2 != nil:
|
||||
fail("préfixe", err2)
|
||||
case err3 != nil:
|
||||
fail("préfixe", err3)
|
||||
case r1 == r1g:
|
||||
// Un moteur qui ignore add_generation_prompt rend l'amorce de réponse
|
||||
// dans les deux cas : la comparaison conclurait à tort « instable ».
|
||||
notes = append(notes, "add_generation_prompt ignoré par le moteur")
|
||||
case !strings.Contains(r1, tplMarkT1) || !strings.Contains(r2, tplMarkB):
|
||||
notes = append(notes, "préfixe : rendu incohérent")
|
||||
case strings.HasPrefix(r2, r1):
|
||||
r.PrefixStable = tplYes
|
||||
default:
|
||||
r.PrefixStable = tplNo
|
||||
}
|
||||
if err1 == nil && !strings.Contains(r1, tplMarkR1) && r.PreservesHistory == tplNo {
|
||||
// Même au dernier tour le raisonnement n'apparaît pas : c'est le moteur
|
||||
// (ou le gabarit) qui ignore reasoning_content, pas un tri par position.
|
||||
notes = append(notes, "reasoning_content jamais rendu")
|
||||
}
|
||||
r.Note = strings.Join(notes, " ; ")
|
||||
r.cacheable = definitive
|
||||
return r
|
||||
}
|
||||
@@ -0,0 +1,432 @@
|
||||
package loki
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"net/url"
|
||||
"os"
|
||||
"regexp"
|
||||
"strings"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// Les faux moteurs des autres tests ne répondent pas tous en JSON sur /props :
|
||||
// la sonde automatique (lancée après chaque complétion du fil principal) y est
|
||||
// coupée, et appelée explicitement ici.
|
||||
func TestMain(m *testing.M) {
|
||||
tplProbeAuto = false
|
||||
os.Exit(m.Run())
|
||||
}
|
||||
|
||||
func freshTplProbe(t *testing.T) {
|
||||
t.Helper()
|
||||
reset := func() {
|
||||
tplProbeMu.Lock()
|
||||
tplProbeCache = map[string]tplProbeResult{}
|
||||
tplProbeLast = nil
|
||||
tplProbeRunning = false
|
||||
tplProbeTried = time.Time{}
|
||||
tplProbeLogged = ""
|
||||
tplProbeMu.Unlock()
|
||||
}
|
||||
reset()
|
||||
old := tplProbeBackoff
|
||||
tplProbeBackoff = time.Millisecond
|
||||
t.Cleanup(func() { reset(); tplProbeBackoff = old })
|
||||
}
|
||||
|
||||
// fakeTpl : un llama-server réduit à /health, /props et /apply-template, avec
|
||||
// un gabarit façon ChatML écrit à la main.
|
||||
//
|
||||
// qwen3 : le raisonnement n'est rendu qu'après le dernier message utilisateur ;
|
||||
// keep : il est toujours rendu ;
|
||||
// noagp : comme keep, mais add_generation_prompt est ignoré (vieux moteur).
|
||||
type fakeTpl struct {
|
||||
mu sync.Mutex
|
||||
mode string
|
||||
key string // Bearer exigé hors /health
|
||||
applyStatus []int // statuts imposés aux prochains /apply-template (0 = rendu)
|
||||
applyBody string
|
||||
paths []string
|
||||
applies []map[string]any
|
||||
}
|
||||
|
||||
var reCallID = regexp.MustCompile(`^[A-Za-z0-9]{9}$`)
|
||||
|
||||
func (f *fakeTpl) render(body map[string]any) (string, int) {
|
||||
msgs, _ := body["messages"].([]any)
|
||||
lastUser := -1
|
||||
for i, m := range msgs {
|
||||
if m.(map[string]any)["role"] == "user" {
|
||||
lastUser = i
|
||||
}
|
||||
}
|
||||
var b strings.Builder
|
||||
if tools, ok := body["tools"].([]any); ok {
|
||||
fmt.Fprintf(&b, "<tools n=%d>\n", len(tools))
|
||||
}
|
||||
for i, raw := range msgs {
|
||||
m := raw.(map[string]any)
|
||||
role, _ := m["role"].(string)
|
||||
b.WriteString("<|im_start|>" + role + "\n")
|
||||
if r, _ := m["reasoning_content"].(string); r != "" && role == "assistant" && (f.mode != "qwen3" || i > lastUser) {
|
||||
b.WriteString("<think>" + r + "</think>")
|
||||
}
|
||||
c, _ := m["content"].(string)
|
||||
b.WriteString(c)
|
||||
calls, _ := m["tool_calls"].([]any)
|
||||
for _, rc := range calls {
|
||||
call := rc.(map[string]any)
|
||||
id, _ := call["id"].(string)
|
||||
// Gabarit strict : identifiant de 9 alphanumériques, résultat qui suit.
|
||||
if !reCallID.MatchString(id) || i+1 >= len(msgs) || msgs[i+1].(map[string]any)["tool_call_id"] != id {
|
||||
return `{"error":{"code":500,"message":"Jinja Exception: tool call id invalide"}}`, 500
|
||||
}
|
||||
fn := call["function"].(map[string]any)
|
||||
b.WriteString("<tool_call>" + fn["name"].(string) + "</tool_call>")
|
||||
}
|
||||
b.WriteString("<|im_end|>\n")
|
||||
}
|
||||
if agp, _ := body["add_generation_prompt"].(bool); agp || f.mode == "noagp" {
|
||||
b.WriteString("<|im_start|>assistant\n")
|
||||
}
|
||||
out, _ := json.Marshal(map[string]string{"prompt": b.String()})
|
||||
return string(out), 200
|
||||
}
|
||||
|
||||
func (f *fakeTpl) handler(w http.ResponseWriter, r *http.Request) {
|
||||
f.mu.Lock()
|
||||
defer f.mu.Unlock()
|
||||
f.paths = append(f.paths, r.URL.Path)
|
||||
if r.URL.Path == "/health" {
|
||||
_, _ = w.Write([]byte(`{"status":"ok"}`))
|
||||
return
|
||||
}
|
||||
if f.key != "" && r.Header.Get("Authorization") != "Bearer "+f.key {
|
||||
w.WriteHeader(http.StatusUnauthorized)
|
||||
_, _ = w.Write([]byte(`{"error":{"code":401,"message":"Invalid API Key"}}`))
|
||||
return
|
||||
}
|
||||
switch r.URL.Path {
|
||||
case "/props":
|
||||
fmt.Fprintf(w, `{"build_info":"b7000-abc","model_path":"/models/Qwen3-8B-Q8_0.gguf","chat_template":%q}`, f.mode)
|
||||
case "/apply-template":
|
||||
var body map[string]any
|
||||
_ = json.NewDecoder(r.Body).Decode(&body)
|
||||
f.applies = append(f.applies, body)
|
||||
if len(f.applyStatus) > 0 {
|
||||
s := f.applyStatus[0]
|
||||
f.applyStatus = f.applyStatus[1:]
|
||||
if s != 0 {
|
||||
w.WriteHeader(s)
|
||||
_, _ = w.Write([]byte(f.applyBody))
|
||||
return
|
||||
}
|
||||
}
|
||||
out, status := f.render(body)
|
||||
w.WriteHeader(status)
|
||||
_, _ = w.Write([]byte(out))
|
||||
default:
|
||||
w.WriteHeader(http.StatusNotFound)
|
||||
}
|
||||
}
|
||||
|
||||
func (f *fakeTpl) start(t *testing.T) {
|
||||
t.Helper()
|
||||
srv := httptest.NewServer(http.HandlerFunc(f.handler))
|
||||
t.Cleanup(srv.Close)
|
||||
u, _ := url.Parse(srv.URL)
|
||||
if err := SetConfigKey("PORT", u.Port()); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
|
||||
func (f *fakeTpl) count(path string) int {
|
||||
f.mu.Lock()
|
||||
defer f.mu.Unlock()
|
||||
n := 0
|
||||
for _, p := range f.paths {
|
||||
if p == path {
|
||||
n++
|
||||
}
|
||||
}
|
||||
return n
|
||||
}
|
||||
|
||||
// Jamais de complétion : avec un seul slot, elle évincerait le cache de la
|
||||
// conversation que la sonde est censée servir.
|
||||
func (f *fakeTpl) assertNoCompletion(t *testing.T) {
|
||||
t.Helper()
|
||||
f.mu.Lock()
|
||||
defer f.mu.Unlock()
|
||||
for _, p := range f.paths {
|
||||
if strings.Contains(p, "completion") {
|
||||
t.Fatalf("la sonde a appelé %s", p)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func probeTools() []Tool {
|
||||
return []Tool{{Type: "function", Function: ToolFunction{Name: "bash", Description: "d", Parameters: map[string]any{"type": "object"}}}}
|
||||
}
|
||||
|
||||
func TestSondeGabaritQwen3(t *testing.T) {
|
||||
testHome(t)
|
||||
freshTplProbe(t)
|
||||
f := &fakeTpl{mode: "qwen3"}
|
||||
f.start(t)
|
||||
kw := map[string]any{"enable_thinking": false}
|
||||
r := tplProbeEnsure(context.Background(), newTplShape(probeTools(), kw, ""))
|
||||
if r.PreservesHistory != tplNo || r.PrefixStable != tplNo {
|
||||
t.Fatalf("Qwen3 retire le raisonnement passé : %+v", r)
|
||||
}
|
||||
if r.Build != "b7000-abc" || r.Model != "Qwen3-8B-Q8_0.gguf" || r.Tools != 1 || !r.cacheable {
|
||||
t.Fatalf("métadonnées : %+v", r)
|
||||
}
|
||||
f.assertNoCompletion(t)
|
||||
// Mêmes outils et chat_template_kwargs que la complétion ; jamais
|
||||
// d'amorce de réponse, sauf le rendu qui vérifie que le moteur l'honore.
|
||||
gen := 0
|
||||
for _, a := range f.applies {
|
||||
if tools, _ := a["tools"].([]any); len(tools) != 1 {
|
||||
t.Fatalf("outils non transmis : %v", a["tools"])
|
||||
}
|
||||
if k, _ := a["chat_template_kwargs"].(map[string]any); k["enable_thinking"] != false {
|
||||
t.Fatalf("chat_template_kwargs : %v", a["chat_template_kwargs"])
|
||||
}
|
||||
if a["add_generation_prompt"] == true {
|
||||
gen++
|
||||
}
|
||||
}
|
||||
if len(f.applies) != 4 || gen != 1 {
|
||||
t.Fatalf("%d rendus dont %d avec amorce", len(f.applies), gen)
|
||||
}
|
||||
// Même moteur, même modèle, même gabarit, même forme : réponse en cache.
|
||||
tplProbeEnsure(context.Background(), newTplShape(probeTools(), kw, ""))
|
||||
if n := f.count("/apply-template"); n != 4 {
|
||||
t.Fatalf("cache ignoré : %d rendus", n)
|
||||
}
|
||||
// Autre forme (raisonnement non coupé) : nouvelle sonde.
|
||||
tplProbeEnsure(context.Background(), newTplShape(probeTools(), nil, ""))
|
||||
if n := f.count("/apply-template"); n != 8 {
|
||||
t.Fatalf("forme différente non resondée : %d rendus", n)
|
||||
}
|
||||
// Exposé dans /api/perf/summary.
|
||||
rec := httptest.NewRecorder()
|
||||
handlePerfSummary(rec, httptest.NewRequest(http.MethodGet, "/api/perf/summary", nil))
|
||||
var sum struct {
|
||||
Template *struct {
|
||||
PreservesHistory string `json:"preserves_history"`
|
||||
PrefixStable string `json:"prefix_stable"`
|
||||
} `json:"template"`
|
||||
}
|
||||
if err := json.Unmarshal(rec.Body.Bytes(), &sum); err != nil || sum.Template == nil || sum.Template.PrefixStable != "no" {
|
||||
t.Fatalf("résumé : %s", rec.Body.String())
|
||||
}
|
||||
if strings.Contains(rec.Body.String(), "loki-sonde") {
|
||||
t.Fatal("un rendu a fuité dans le résumé")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSondeGabaritConserveLHistorique(t *testing.T) {
|
||||
testHome(t)
|
||||
freshTplProbe(t)
|
||||
f := &fakeTpl{mode: "keep"}
|
||||
f.start(t)
|
||||
r := tplProbeEnsure(context.Background(), newTplShape(nil, nil, ""))
|
||||
if r.PreservesHistory != tplYes || r.PrefixStable != tplYes {
|
||||
t.Fatalf("gabarit qui garde tout : %+v", r)
|
||||
}
|
||||
for _, a := range f.applies {
|
||||
if _, ok := a["tools"]; ok {
|
||||
t.Fatal("outils envoyés alors que la complétion n'en avait pas")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Un moteur qui ignore add_generation_prompt ne doit pas faire conclure
|
||||
// « instable » : l'amorce rendue dans les deux cas fausserait la comparaison.
|
||||
func TestSondeGabaritAmorceIgnoree(t *testing.T) {
|
||||
testHome(t)
|
||||
freshTplProbe(t)
|
||||
f := &fakeTpl{mode: "noagp"}
|
||||
f.start(t)
|
||||
r := tplProbeEnsure(context.Background(), newTplShape(nil, nil, ""))
|
||||
if r.PrefixStable != tplUnknown || r.PreservesHistory != tplYes {
|
||||
t.Fatalf("amorce ignorée : %+v", r)
|
||||
}
|
||||
}
|
||||
|
||||
// Moteur ancien ou fork sans la route : inconnu, rangé (ça ne changera pas
|
||||
// avant un autre binaire), et aucun repli sur une complétion.
|
||||
func TestSondeGabarit404(t *testing.T) {
|
||||
testHome(t)
|
||||
freshTplProbe(t)
|
||||
f := &fakeTpl{mode: "qwen3", applyStatus: []int{404, 404, 404, 404}, applyBody: "Not Found"}
|
||||
f.start(t)
|
||||
r := tplProbeEnsure(context.Background(), newTplShape(probeTools(), nil, ""))
|
||||
if r.PreservesHistory != tplUnknown || r.PrefixStable != tplUnknown || !r.cacheable {
|
||||
t.Fatalf("404 : %+v", r)
|
||||
}
|
||||
f.assertNoCompletion(t)
|
||||
tplProbeEnsure(context.Background(), newTplShape(probeTools(), nil, ""))
|
||||
if n := f.count("/apply-template"); n != 2 {
|
||||
t.Fatalf("404 resondé : %d rendus", n)
|
||||
}
|
||||
}
|
||||
|
||||
// Gabarit qui lève une exception (raise_exception) : inconnu, jamais oui/non.
|
||||
func TestSondeGabaritExceptionDuGabarit(t *testing.T) {
|
||||
testHome(t)
|
||||
freshTplProbe(t)
|
||||
f := &fakeTpl{mode: "qwen3", applyStatus: []int{500, 500, 500, 500},
|
||||
applyBody: `{"error":{"code":500,"message":"Jinja Exception: Conversation roles must alternate"}}`}
|
||||
f.start(t)
|
||||
r := tplProbeEnsure(context.Background(), newTplShape(probeTools(), nil, ""))
|
||||
if r.PreservesHistory != tplUnknown || r.PrefixStable != tplUnknown || !strings.Contains(r.Note, "500") {
|
||||
t.Fatalf("exception du gabarit : %+v", r)
|
||||
}
|
||||
f.assertNoCompletion(t)
|
||||
}
|
||||
|
||||
// Moteur protégé par API_KEY : la sonde s'authentifie comme les autres appels
|
||||
// internes. Sans clé, inconnu — et pas rangé : la clé peut arriver.
|
||||
func TestSondeGabarit401(t *testing.T) {
|
||||
testHome(t)
|
||||
freshTplProbe(t)
|
||||
f := &fakeTpl{mode: "keep", key: "sk-loki-test"}
|
||||
f.start(t)
|
||||
r := tplProbeEnsure(context.Background(), newTplShape(nil, nil, ""))
|
||||
if r.PreservesHistory != tplUnknown || r.PrefixStable != tplUnknown || !strings.Contains(r.Note, "401") {
|
||||
t.Fatalf("401 : %+v", r)
|
||||
}
|
||||
if err := writeAPIKey("sk-loki-test"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
r = tplProbeEnsure(context.Background(), newTplShape(nil, nil, ""))
|
||||
if r.PreservesHistory != tplYes || r.PrefixStable != tplYes {
|
||||
t.Fatalf("avec la clé : %+v", r)
|
||||
}
|
||||
}
|
||||
|
||||
// Modèle en chargement : les 503 sont relancés, sans rien conclure entre-temps.
|
||||
func TestSondeGabarit503Relance(t *testing.T) {
|
||||
testHome(t)
|
||||
freshTplProbe(t)
|
||||
f := &fakeTpl{mode: "qwen3", applyStatus: []int{503, 503}, applyBody: `{"error":{"code":503,"message":"Loading model"}}`}
|
||||
f.start(t)
|
||||
r := tplProbeEnsure(context.Background(), newTplShape(nil, nil, ""))
|
||||
if r.PreservesHistory != tplNo || r.PrefixStable != tplNo {
|
||||
t.Fatalf("après 503 : %+v", r)
|
||||
}
|
||||
// 503 persistant : inconnu, non rangé.
|
||||
freshTplProbe(t)
|
||||
f.mu.Lock()
|
||||
f.applyStatus = []int{503, 503, 503, 503, 503, 503, 503, 503, 503, 503, 503, 503}
|
||||
f.mu.Unlock()
|
||||
r = tplProbeEnsure(context.Background(), newTplShape(nil, nil, ""))
|
||||
if r.PreservesHistory != tplUnknown || r.cacheable {
|
||||
t.Fatalf("503 persistant : %+v", r)
|
||||
}
|
||||
}
|
||||
|
||||
// Preset externe : aucune requête, nulle part, et rien d'affiché.
|
||||
func TestSondeGabaritPresetExterne(t *testing.T) {
|
||||
testHome(t)
|
||||
freshTplProbe(t)
|
||||
f := &fakeTpl{mode: "keep"}
|
||||
srv := httptest.NewServer(http.HandlerFunc(f.handler))
|
||||
t.Cleanup(srv.Close)
|
||||
if err := WriteConfig(parseEnv(externalPresetContent(srv.URL, "gpt-4o-mini", "", "", false))); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
u, _ := url.Parse(srv.URL)
|
||||
if err := SetConfigKey("PORT", u.Port()); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
r := tplProbeEnsure(context.Background(), newTplShape(nil, nil, ""))
|
||||
if r.PreservesHistory != tplUnknown || r.PrefixStable != tplUnknown {
|
||||
t.Fatalf("preset externe : %+v", r)
|
||||
}
|
||||
if n := len(f.paths); n != 0 {
|
||||
t.Fatalf("preset externe sondé : %v", f.paths)
|
||||
}
|
||||
if _, ok := tplCapsCurrent(); ok {
|
||||
t.Fatal("verdict affiché pour un preset externe")
|
||||
}
|
||||
}
|
||||
|
||||
// Bout en bout : une complétion du fil principal déclenche la sonde en tâche
|
||||
// de fond, avec les outils et kwargs de CETTE requête, sans rien ajouter à la
|
||||
// requête de complétion.
|
||||
func TestSondeGabaritDeclencheeParRunChat(t *testing.T) {
|
||||
testHome(t)
|
||||
freshTplProbe(t)
|
||||
freshPerf(t)
|
||||
tplProbeAuto = true
|
||||
t.Cleanup(func() { tplProbeAuto = false })
|
||||
if err := SetConfigKey("REASONING", "off"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
f := &fakeTpl{mode: "qwen3"}
|
||||
var mu sync.Mutex
|
||||
var chat map[string]any
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
if r.URL.Path == "/v1/chat/completions" {
|
||||
mu.Lock()
|
||||
_ = json.NewDecoder(r.Body).Decode(&chat)
|
||||
mu.Unlock()
|
||||
w.Header().Set("Content-Type", "text/event-stream")
|
||||
_, _ = w.Write([]byte(sseChunk("ok") +
|
||||
`data: {"choices":[{"delta":{},"finish_reason":"stop"}],"usage":{"prompt_tokens":50,"completion_tokens":1},"timings":{"prompt_n":50,"cache_n":0}}` + "\n\n" +
|
||||
"data: [DONE]\n\n"))
|
||||
return
|
||||
}
|
||||
f.handler(w, r)
|
||||
}))
|
||||
t.Cleanup(srv.Close)
|
||||
u, _ := url.Parse(srv.URL)
|
||||
if err := SetConfigKey("PORT", u.Port()); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
tools := probeTools()
|
||||
if _, err := runChatTools(context.Background(), []Message{{Role: "user", Content: "salut"}}, tools, 0.7, Caps{}, func(StreamEvent) bool { return true }); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
deadline := time.Now().Add(5 * time.Second)
|
||||
var r tplProbeResult
|
||||
for {
|
||||
var ok bool
|
||||
if r, ok = tplCapsCurrent(); ok || time.Now().After(deadline) {
|
||||
break
|
||||
}
|
||||
time.Sleep(10 * time.Millisecond)
|
||||
}
|
||||
if r.PrefixStable != tplNo || r.Tools != 1 || r.Kwargs["enable_thinking"] != false {
|
||||
t.Fatalf("sonde après complétion : %+v", r)
|
||||
}
|
||||
f.mu.Lock()
|
||||
for _, a := range f.applies {
|
||||
mu.Lock()
|
||||
same := fmt.Sprint(a["tools"]) == fmt.Sprint(chat["tools"]) && fmt.Sprint(a["chat_template_kwargs"]) == fmt.Sprint(chat["chat_template_kwargs"])
|
||||
mu.Unlock()
|
||||
if !same {
|
||||
f.mu.Unlock()
|
||||
t.Fatalf("forme différente de la complétion :\nsonde %v / %v\nchat %v / %v", a["tools"], a["chat_template_kwargs"], chat["tools"], chat["chat_template_kwargs"])
|
||||
}
|
||||
}
|
||||
f.mu.Unlock()
|
||||
// La sonde ne laisse aucune trace dans la requête de complétion.
|
||||
mu.Lock()
|
||||
defer mu.Unlock()
|
||||
for k := range chat {
|
||||
if strings.Contains(k, "template") && k != "chat_template_kwargs" || k == "add_generation_prompt" {
|
||||
t.Fatalf("clé %q ajoutée à la complétion", k)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1869,6 +1869,15 @@ func runChatTools(ctx context.Context, messages []Message, tools []Tool, tempera
|
||||
s.LostAlert = perfAlert(rec, perfLostAlertAt(chatCfg))
|
||||
}
|
||||
cb(StreamEvent{Stats: &s})
|
||||
// Sonde de gabarit (chat_tplprobe.go), en tâche de fond et après coup :
|
||||
// la complétion est déjà lue, la sonde ne retarde aucun jeton. Fil
|
||||
// principal seulement, sur une requête de forme normale (pas une
|
||||
// relance outils coupés ou neutralisés), pour reproduire exactement
|
||||
// les outils et chat_template_kwargs d'un vrai tour. Rien n'en revient
|
||||
// dans la requête suivante.
|
||||
if !ep.External && ptag.kind == perfMain && !disableTools && !toolChoiceNone {
|
||||
tplProbeKick(tools, reasoningKwargs, reasoningEffort)
|
||||
}
|
||||
if !ep.External {
|
||||
peakGen = max(peakGen, stats.GenTokens)
|
||||
if lastEst > 0 && stats.PromptTokensTotal > 0 {
|
||||
|
||||
@@ -472,6 +472,9 @@ type perfSummary struct {
|
||||
DraftN int `json:"draft_n"`
|
||||
DraftAccepted int `json:"draft_accepted"`
|
||||
DraftRate *float64 `json:"draft_rate,omitempty"`
|
||||
// Template : dernier verdict de la sonde de gabarit (chat_tplprobe.go),
|
||||
// absent tant qu'aucune n'a tourné ou sur un preset externe.
|
||||
Template *tplProbeResult `json:"template,omitempty"`
|
||||
}
|
||||
|
||||
var perfDepthBuckets = []struct {
|
||||
@@ -580,5 +583,9 @@ func handlePerfSummary(w http.ResponseWriter, r *http.Request) {
|
||||
sendJSON(w, http.StatusMethodNotAllowed, map[string]any{"error": "GET seulement"})
|
||||
return
|
||||
}
|
||||
sendJSON(w, http.StatusOK, perfSummarize(perfLog.snapshot()))
|
||||
s := perfSummarize(perfLog.snapshot())
|
||||
if r, ok := tplCapsCurrent(); ok {
|
||||
s.Template = &r
|
||||
}
|
||||
sendJSON(w, http.StatusOK, s)
|
||||
}
|
||||
Reference in new issue
Block a user