mirror of
https://github.com/R0m1k3/Loki.git
synced 2026-10-11 17:26:57 +02:00
Bench : une mesure honnête à profondeur réelle, en tâche de fond
L'ancien bench ne lisait pas le statut HTTP, inventait un partage 15/85 du temps quand les timings manquaient (et l'enregistrait comme mesuré), tuilait un petit corpus qu'un brouillon recopiait, et ne disait rien du prefill à 30k ni du chemin où le cache est repris. Synchrone, il était coupé par les proxys (60-100 s) pendant que le moteur continuait, et un simple GET suffisait à le lancer. - Tâche de fond : POST /api/bench (202, 405 sur GET, 409 si occupé), /api/bench/status pour la progression, /api/bench/cancel ; chaque requête au moteur porte le contexte annulable. - Verrou de génération tenu pendant toute la mesure : chat, tâches et compaction reçoivent un refus clair ; les clients /v1 un 503 avec Retry-After ; /slots occupé (client externe, CLI) refuse aussi. - Mode rapide par défaut : préchauffage jeté + la ligne courte, sur le chat avec gabarit, raisonnement et échantillonnage du preset, seed fixe. - Mode complet (bouton « bench complet », loki bench --full) : prefill à froid à D = min(CTX/2, 32k, ce qui laisse tenir les tours), 16k si des poids tournent sur CPU, puis 3 tours user → assistant → user de code jamais vu. Reprise du cache lue dans cache_n, jamais supposée ; note pour les hybrides (points de contrôle). Pas d'ignore_eos. - Rien n'est enregistré sans réponses 200 et timings réels, ni pour un résultat partiel (budget de 6 min par requête, contexte plein). - Empreinte enregistrée (configuration + CUDA_VISIBLE_DEVICES + protocole) ; les anciennes mesures retombent sur le nom du modèle. Types de cache KV et build du moteur affichés, KV quantifié signalé (zone grise). - Linux : indication « possible thrash » (read_bytes, VmRSS), jamais enregistrée. - Effacement du slot en fin de bench inchangé (engineSideJob, différé : il tourne aussi sur erreur et annulation). Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
1 parent
9f6bd725e6
commit
cf83f5f629
17 files changed
+1904
-245
No files matched your search
@@ -230,7 +230,7 @@ func TestPresetExterneVisionEtBench(t *testing.T) {
|
||||
if !visionEnabled() || !engineSeesImages() {
|
||||
t.Error("preset externe déclaré multimodal : la vision devrait être active")
|
||||
}
|
||||
if _, err := runBench(10, 10); err == nil {
|
||||
if _, err := runBench(context.Background(), benchOpts{Prompt: 10, Predict: 10}, nil); err == nil {
|
||||
t.Error("benchmark lancé sur un preset externe")
|
||||
}
|
||||
if got := chatModelName(); got != "gpt-4o" {
|
||||
|
||||
@@ -77,6 +77,9 @@ type Conversation struct {
|
||||
// à une génération fantôme dans la discussion ouverte.
|
||||
runningTaskID string
|
||||
runningTaskName string
|
||||
// benching : le verrou est tenu par un benchmark (llm_bench_job.go). Même
|
||||
// logique que la tâche : l'interface et les refus disent qui occupe le moteur.
|
||||
benching bool
|
||||
|
||||
// File d'attente des messages envoyés PENDANT une génération (AJEAN
|
||||
// 0.14.0). Ils sont injectés dans le tour en cours à la prochaine frontière
|
||||
@@ -443,8 +446,11 @@ var ErrBusy = fmt.Errorf("génération en cours")
|
||||
// nomme la tâche et on dit comment reprendre la main.
|
||||
func (c *Conversation) busyReason() error {
|
||||
c.mu.Lock()
|
||||
name := c.runningTaskName
|
||||
name, bench := c.runningTaskName, c.benching
|
||||
c.mu.Unlock()
|
||||
if bench {
|
||||
return errBenchBusy
|
||||
}
|
||||
if name == "" {
|
||||
return ErrBusy
|
||||
}
|
||||
@@ -550,8 +556,8 @@ func (c *Conversation) seenCIDLocked(cid string) bool {
|
||||
// EnqueueOrStart démarre un tour tout de suite si le moteur est libre, sinon
|
||||
// MET EN FILE le message (AJEAN 0.14.0) — au lieu du 409 d'avant, qui obligeait
|
||||
// à arrêter la réponse pour ajouter une précision. Seul un tour de CHAT accepte
|
||||
// une file : une tâche planifiée qui occupe le modèle garde le refus
|
||||
// (busyReason), sa fin ne dépile rien. cid = identifiant de l'envoi (voir
|
||||
// une file : une tâche planifiée ou un benchmark qui occupe le modèle garde le
|
||||
// refus (busyReason), sa fin ne dépile rien. cid = identifiant de l'envoi (voir
|
||||
// recentCIDs) ; un doublon renvoie ErrDupSend.
|
||||
func (c *Conversation) EnqueueOrStart(cid, text string, files []attachInfo, caps Caps, temperature float64) (bool, error) {
|
||||
c.mu.Lock()
|
||||
@@ -559,7 +565,7 @@ func (c *Conversation) EnqueueOrStart(cid, text string, files []attachInfo, caps
|
||||
c.mu.Unlock()
|
||||
return false, ErrDupSend
|
||||
}
|
||||
if c.Generating && c.runningTaskName == "" {
|
||||
if c.Generating && c.runningTaskName == "" && !c.benching {
|
||||
c.queued = append(c.queued, queuedMsg{text: text, files: files, caps: caps, temp: temperature})
|
||||
c.mu.Unlock()
|
||||
return true, nil
|
||||
@@ -571,7 +577,7 @@ func (c *Conversation) EnqueueOrStart(cid, text string, files []attachInfo, caps
|
||||
// appareils qui envoient en même temps) : en file plutôt qu'un refus
|
||||
// (AJEAN 0.17.4). Une tâche planifiée garde le refus : sa fin ne dépile rien.
|
||||
c.mu.Lock()
|
||||
if c.Generating && c.runningTaskName == "" {
|
||||
if c.Generating && c.runningTaskName == "" && !c.benching {
|
||||
c.queued = append(c.queued, queuedMsg{text: text, files: files, caps: caps, temp: temperature})
|
||||
c.mu.Unlock()
|
||||
// Ce tour a pu finir entre-temps : sans relance, le message attendrait
|
||||
|
||||
+865
-127
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,235 @@
|
||||
package loki
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
)
|
||||
|
||||
// Le benchmark tourne en TÂCHE DE FOND. Le mode complet dure de une à cinq
|
||||
// minutes : une requête HTTP synchrone était coupée bien avant par un reverse
|
||||
// proxy (nginx d'Unraid à 60 s, Cloudflare à 100 s) ou par le délai de
|
||||
// l'interface (30 s), pendant que le moteur, lui, continuait — et un second
|
||||
// clic empilait un deuxième bench derrière le premier. Désormais :
|
||||
//
|
||||
// POST /api/bench lance (202 + id), 409 si le moteur est occupé
|
||||
// GET /api/bench/status phase en cours, puis résultat ou erreur
|
||||
// POST /api/bench/cancel annule (chaque requête au moteur porte le contexte)
|
||||
// GET /api/bench/last dernier résultat enregistré
|
||||
//
|
||||
// Pendant toute la mesure, le bench tient le verrou de génération de la
|
||||
// conversation : un message, une tâche planifiée ou une compaction reçoivent
|
||||
// un refus clair au lieu d'attendre des minutes dans la file du moteur. Les
|
||||
// clients /v1 reçoivent un 503 avec Retry-After (llm_oai.go). Restent hors du
|
||||
// verrou les appels annexes déclenchés par la fin d'un tour (titre, mémoire) :
|
||||
// le verrou les précède de toute façon.
|
||||
|
||||
var errBenchBusy = errors.New("un benchmark occupe le moteur — attends sa fin ou annule-le")
|
||||
|
||||
// benchLease prend le verrou de génération pour un benchmark. cancel est
|
||||
// branché sur le bouton stop du chat, comme pour une tâche planifiée. La
|
||||
// fonction rendue le libère ; si un Reset l'a déjà rendu entre-temps (epoch
|
||||
// changé), elle ne touche pas au tour qui a pu démarrer depuis.
|
||||
func (c *Conversation) benchLease(cancel context.CancelFunc) (func(), error) {
|
||||
c.mu.Lock()
|
||||
defer c.mu.Unlock()
|
||||
if c.Generating {
|
||||
return nil, ErrBusy
|
||||
}
|
||||
c.Generating, c.benching, c.cancel = true, true, cancel
|
||||
epoch := c.epoch
|
||||
var once sync.Once
|
||||
return func() {
|
||||
once.Do(func() {
|
||||
c.mu.Lock()
|
||||
c.benching = false
|
||||
if c.epoch == epoch {
|
||||
c.Generating = false
|
||||
c.cancel = nil
|
||||
}
|
||||
c.mu.Unlock()
|
||||
})
|
||||
}, nil
|
||||
}
|
||||
|
||||
// benchJob : le benchmark en cours ou le dernier terminé (un seul à la fois).
|
||||
var benchJob struct {
|
||||
mu sync.Mutex
|
||||
seq int
|
||||
running bool
|
||||
id string
|
||||
mode string
|
||||
phase string
|
||||
step int
|
||||
steps int
|
||||
started time.Time
|
||||
finished time.Time
|
||||
cancel context.CancelFunc
|
||||
result *benchResult
|
||||
err string
|
||||
canceled bool
|
||||
}
|
||||
|
||||
// benchRunning : un benchmark de ce processus est-il en cours ?
|
||||
func benchRunning() bool {
|
||||
benchJob.mu.Lock()
|
||||
defer benchJob.mu.Unlock()
|
||||
return benchJob.running
|
||||
}
|
||||
|
||||
// benchRunner est runBench, remplaçable dans les tests.
|
||||
var benchRunner = runBench
|
||||
|
||||
// benchStart lance un benchmark en arrière-plan. code : le statut HTTP à
|
||||
// renvoyer en cas de refus.
|
||||
func benchStart(opts benchOpts) (id string, code int, err error) {
|
||||
// Hors verrou : healthCheck peut attendre le moteur quelques secondes, le
|
||||
// suivi de progression ne doit pas rester bloqué derrière.
|
||||
if externalActive() {
|
||||
return "", http.StatusBadRequest, errors.New("benchmark indisponible : le preset actif est une API externe")
|
||||
}
|
||||
if !healthCheck() {
|
||||
return "", http.StatusServiceUnavailable, errModelLoading
|
||||
}
|
||||
benchJob.mu.Lock()
|
||||
defer benchJob.mu.Unlock()
|
||||
if benchJob.running {
|
||||
return "", http.StatusConflict, errors.New("un benchmark tourne déjà")
|
||||
}
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
release, err := conv.benchLease(cancel)
|
||||
if err != nil {
|
||||
cancel()
|
||||
return "", http.StatusConflict, conv.busyReason()
|
||||
}
|
||||
benchJob.seq++
|
||||
id = fmt.Sprintf("b%d-%d", time.Now().Unix(), benchJob.seq)
|
||||
benchJob.running, benchJob.id, benchJob.mode = true, id, opts.Mode
|
||||
benchJob.phase, benchJob.step, benchJob.steps = "démarrage", 0, 0
|
||||
benchJob.started, benchJob.finished = time.Now(), time.Time{}
|
||||
benchJob.cancel, benchJob.result, benchJob.err, benchJob.canceled = cancel, nil, "", false
|
||||
go func() {
|
||||
defer cancel()
|
||||
defer release()
|
||||
res, err := benchRunner(ctx, opts, func(phase string, step, steps int) {
|
||||
benchJob.mu.Lock()
|
||||
if benchJob.id == id {
|
||||
benchJob.phase, benchJob.step, benchJob.steps = phase, step, steps
|
||||
}
|
||||
benchJob.mu.Unlock()
|
||||
})
|
||||
benchJob.mu.Lock()
|
||||
defer benchJob.mu.Unlock()
|
||||
if benchJob.id != id {
|
||||
return
|
||||
}
|
||||
benchJob.running, benchJob.finished, benchJob.cancel = false, time.Now(), nil
|
||||
benchJob.result = res
|
||||
if err != nil {
|
||||
benchJob.canceled = ctx.Err() != nil
|
||||
benchJob.err = err.Error()
|
||||
if benchJob.canceled {
|
||||
benchJob.err = "benchmark annulé"
|
||||
}
|
||||
}
|
||||
}()
|
||||
return id, http.StatusAccepted, nil
|
||||
}
|
||||
|
||||
// benchCancel annule le benchmark en cours. false : aucun ne tournait.
|
||||
func benchCancel() bool {
|
||||
benchJob.mu.Lock()
|
||||
defer benchJob.mu.Unlock()
|
||||
if !benchJob.running || benchJob.cancel == nil {
|
||||
return false
|
||||
}
|
||||
benchJob.cancel()
|
||||
return true
|
||||
}
|
||||
|
||||
// benchStatus : l'état à renvoyer à l'interface.
|
||||
func benchStatus() map[string]any {
|
||||
benchJob.mu.Lock()
|
||||
defer benchJob.mu.Unlock()
|
||||
st := map[string]any{"ok": true, "running": benchJob.running, "id": benchJob.id, "mode": benchJob.mode}
|
||||
if benchJob.id == "" {
|
||||
return st
|
||||
}
|
||||
st["phase"], st["step"], st["steps"] = benchJob.phase, benchJob.step, benchJob.steps
|
||||
end := benchJob.finished
|
||||
if benchJob.running {
|
||||
end = time.Now()
|
||||
}
|
||||
st["elapsed_sec"] = end.Sub(benchJob.started).Seconds()
|
||||
if !benchJob.running {
|
||||
if benchJob.result != nil {
|
||||
st["result"] = benchJob.result
|
||||
st["saved"] = benchSavable(benchJob.result)
|
||||
}
|
||||
if benchJob.err != "" {
|
||||
st["error"] = benchJob.err
|
||||
st["canceled"] = benchJob.canceled
|
||||
}
|
||||
}
|
||||
return st
|
||||
}
|
||||
|
||||
// handleBench lance un benchmark : POST seulement — il occupe le moteur des
|
||||
// minutes durant, une balise <img> sur une page tierce ne doit pas y suffire.
|
||||
// Corps : {"mode":"quick"|"full", "prompt":N, "n":N}, tous facultatifs.
|
||||
func handleBench(w http.ResponseWriter, r *http.Request) {
|
||||
if !postOnly(w, r) {
|
||||
return
|
||||
}
|
||||
var body struct {
|
||||
Mode string `json:"mode"`
|
||||
Prompt int `json:"prompt"`
|
||||
N int `json:"n"`
|
||||
}
|
||||
_ = json.NewDecoder(r.Body).Decode(&body)
|
||||
opts := benchOpts{Mode: benchModeQuick, Prompt: 2000, Predict: 300}
|
||||
if strings.EqualFold(strings.TrimSpace(body.Mode), benchModeFull) {
|
||||
opts.Mode = benchModeFull
|
||||
}
|
||||
if body.Prompt > 0 {
|
||||
opts.Prompt = min(body.Prompt, 32768)
|
||||
}
|
||||
if body.N > 0 {
|
||||
opts.Predict = min(body.N, 4096)
|
||||
}
|
||||
id, code, err := benchStart(opts)
|
||||
if err != nil {
|
||||
sendJSON(w, code, map[string]any{"ok": false, "error": err.Error()})
|
||||
return
|
||||
}
|
||||
sendJSON(w, code, map[string]any{"ok": true, "id": id})
|
||||
}
|
||||
|
||||
// handleBenchStatus : progression du benchmark en cours, ou issue du dernier.
|
||||
func handleBenchStatus(w http.ResponseWriter, r *http.Request) {
|
||||
sendJSON(w, 200, benchStatus())
|
||||
}
|
||||
|
||||
// handleBenchCancel annule le benchmark en cours.
|
||||
func handleBenchCancel(w http.ResponseWriter, r *http.Request) {
|
||||
if !postOnly(w, r) {
|
||||
return
|
||||
}
|
||||
sendJSON(w, 200, map[string]any{"ok": benchCancel()})
|
||||
}
|
||||
|
||||
// handleBenchLast returns the most recent persisted benchmark, or {ok:false}
|
||||
// when none has been run yet.
|
||||
func handleBenchLast(w http.ResponseWriter, r *http.Request) {
|
||||
sb := loadLastBench()
|
||||
if sb == nil {
|
||||
sendJSON(w, 200, map[string]any{"ok": false})
|
||||
return
|
||||
}
|
||||
sendJSON(w, 200, map[string]any{"ok": true, "result": sb.Result, "model": sb.Model, "at": sb.At})
|
||||
}
|
||||
@@ -5,13 +5,13 @@ import "testing"
|
||||
// Un bench n'est affiché que pour le modèle sur lequel il a été mesuré.
|
||||
func TestBenchMatchesPreset(t *testing.T) {
|
||||
sb := savedBench{Model: "Qwen3-27B-Q4_K_M.gguf"}
|
||||
if !benchMatchesPreset(sb, map[string]string{"MODEL": "/models/Qwen3-27B-Q4_K_M.gguf"}) {
|
||||
if !benchMatchesPreset(sb, map[string]string{"MODEL": "/models/Qwen3-27B-Q4_K_M.gguf"}, nil) {
|
||||
t.Fatal("même modèle : le bench doit s'afficher")
|
||||
}
|
||||
if benchMatchesPreset(sb, map[string]string{"MODEL": "/models/Autre.gguf"}) {
|
||||
if benchMatchesPreset(sb, map[string]string{"MODEL": "/models/Autre.gguf"}, nil) {
|
||||
t.Fatal("autre modèle : le bench ne doit pas s'afficher")
|
||||
}
|
||||
if benchMatchesPreset(sb, map[string]string{}) {
|
||||
if benchMatchesPreset(sb, map[string]string{}, nil) {
|
||||
t.Fatal("preset sans modèle : pas de bench")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,543 @@
|
||||
package loki
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"net/url"
|
||||
"strings"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
"unicode/utf8"
|
||||
)
|
||||
|
||||
// fakeBenchEngine simule llama-server pour le protocole du bench : un compte
|
||||
// de jetons à 4 octets par jeton, un cache qui reprend le prompt précédent
|
||||
// quand cache_prompt est vrai, et le journal des requêtes de chat reçues.
|
||||
type fakeBenchEngine struct {
|
||||
mu sync.Mutex
|
||||
nCtx int
|
||||
busy bool
|
||||
failAt int // n° (1-based) de la requête de chat qui répond 500 ; 0 = aucune
|
||||
noTiming bool // réponses sans timings
|
||||
slowHot time.Duration // délai des tours qui reprennent le cache
|
||||
prev int
|
||||
calls []map[string]any
|
||||
}
|
||||
|
||||
func (f *fakeBenchEngine) handler(t *testing.T) http.Handler {
|
||||
return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
switch r.URL.Path {
|
||||
case "/health":
|
||||
w.Write([]byte(`{"status":"ok"}`))
|
||||
case "/slots":
|
||||
fmt.Fprintf(w, `[{"id":0,"is_processing":%v}]`, f.busy)
|
||||
case "/props":
|
||||
fmt.Fprintf(w, `{"build_info":"b9999-test","total_slots":1,"default_generation_settings":{"n_ctx":%d}}`, f.nCtx)
|
||||
case "/tokenize":
|
||||
var b struct {
|
||||
Content string `json:"content"`
|
||||
}
|
||||
json.NewDecoder(r.Body).Decode(&b)
|
||||
toks := make([]int, len(b.Content)/4)
|
||||
json.NewEncoder(w).Encode(map[string]any{"tokens": toks})
|
||||
case "/v1/chat/completions":
|
||||
var p map[string]any
|
||||
if err := json.NewDecoder(r.Body).Decode(&p); err != nil {
|
||||
t.Errorf("corps illisible : %v", err)
|
||||
}
|
||||
f.mu.Lock()
|
||||
f.calls = append(f.calls, p)
|
||||
n := len(f.calls)
|
||||
f.mu.Unlock()
|
||||
if f.failAt == n {
|
||||
http.Error(w, `{"error":"boom"}`, http.StatusInternalServerError)
|
||||
return
|
||||
}
|
||||
cache, _ := p["cache_prompt"].(bool)
|
||||
if d := f.slowHot; cache && d > 0 {
|
||||
select {
|
||||
case <-time.After(d):
|
||||
case <-r.Context().Done():
|
||||
return
|
||||
}
|
||||
}
|
||||
total := 0
|
||||
for _, m := range p["messages"].([]any) {
|
||||
c, _ := m.(map[string]any)["content"].(string)
|
||||
total += len(c)/4 + 4
|
||||
}
|
||||
cached := 0
|
||||
f.mu.Lock()
|
||||
if cache {
|
||||
cached = min(f.prev, total)
|
||||
}
|
||||
f.prev = total
|
||||
f.mu.Unlock()
|
||||
gen := int(p["max_tokens"].(float64))
|
||||
resp := map[string]any{
|
||||
"choices": []any{map[string]any{"message": map[string]any{"role": "assistant", "content": "réponse"}}},
|
||||
"usage": map[string]any{"prompt_tokens": total, "completion_tokens": gen},
|
||||
}
|
||||
if !f.noTiming {
|
||||
resp["timings"] = map[string]any{
|
||||
"prompt_n": total - cached, "prompt_ms": float64(total-cached) / 2, "prompt_per_second": 2000.0,
|
||||
"predicted_n": gen, "predicted_ms": float64(gen) * 20, "predicted_per_second": 50.0,
|
||||
"cache_n": cached, "draft_n": 10, "draft_n_accepted": 7,
|
||||
}
|
||||
}
|
||||
json.NewEncoder(w).Encode(resp)
|
||||
default:
|
||||
http.NotFound(w, r)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func (f *fakeBenchEngine) start(t *testing.T) (benchEngine, *httptest.Server) {
|
||||
srv := httptest.NewServer(f.handler(t))
|
||||
t.Cleanup(srv.Close)
|
||||
return benchEngine{base: srv.URL, client: srv.Client()}, srv
|
||||
}
|
||||
|
||||
// testBenchCorpora : un code de 200 Ko sans aucune ligne répétée.
|
||||
func testBenchCorpora() benchCorpora {
|
||||
var b strings.Builder
|
||||
for i := 0; b.Len() < 200<<10; i++ {
|
||||
fmt.Fprintf(&b, "const ligne%06d = « contenu distinct %d » ;\n", i, i*7)
|
||||
}
|
||||
return benchCorpora{prose: benchCorpus, code: b.String()}
|
||||
}
|
||||
|
||||
func testBenchSetup() benchSetup {
|
||||
return benchSetupFrom(map[string]string{"TOP_K": "40", "MIN_P": "0.05", "REASONING": "off"}, nil)
|
||||
}
|
||||
|
||||
// Le protocole complet, requête par requête : préchauffage, ligne courte et
|
||||
// prefill à froid sans cache, puis trois tours qui prolongent la MÊME
|
||||
// discussion avec le cache — et l'échantillonnage du preset partout.
|
||||
func TestBenchProtocoleComplet(t *testing.T) {
|
||||
f := &fakeBenchEngine{nCtx: 16384}
|
||||
eng, _ := f.start(t)
|
||||
var phases []string
|
||||
res, err := benchRun(context.Background(), eng, benchOpts{Mode: benchModeFull}, testBenchSetup(), testBenchCorpora(),
|
||||
func(phase string, step, steps int) {
|
||||
phases = append(phases, fmt.Sprintf("%d/%d %s", step, steps, phase))
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(f.calls) != 6 {
|
||||
t.Fatalf("%d requêtes de chat, attendu 6 (préchauffage, ligne courte, froid, 3 tours)", len(f.calls))
|
||||
}
|
||||
wantCache := []bool{false, false, false, true, true, true}
|
||||
wantMax := []float64{benchWarmGen, 300, benchGenTokens, benchGenTokens, benchGenTokens, benchGenTokens}
|
||||
wantMsgs := []int{1, 1, 1, 3, 5, 7}
|
||||
for i, p := range f.calls {
|
||||
if p["cache_prompt"] != wantCache[i] {
|
||||
t.Errorf("requête %d : cache_prompt=%v, attendu %v", i+1, p["cache_prompt"], wantCache[i])
|
||||
}
|
||||
if p["max_tokens"] != wantMax[i] {
|
||||
t.Errorf("requête %d : max_tokens=%v, attendu %v", i+1, p["max_tokens"], wantMax[i])
|
||||
}
|
||||
msgs := p["messages"].([]any)
|
||||
if len(msgs) != wantMsgs[i] {
|
||||
t.Errorf("requête %d : %d messages, attendu %d", i+1, len(msgs), wantMsgs[i])
|
||||
}
|
||||
for j, m := range msgs {
|
||||
want := "user"
|
||||
if j%2 == 1 {
|
||||
want = "assistant"
|
||||
}
|
||||
if role := m.(map[string]any)["role"]; role != want {
|
||||
t.Errorf("requête %d, message %d : rôle %v, attendu %s", i+1, j, role, want)
|
||||
}
|
||||
}
|
||||
if p["top_k"] != float64(40) || p["min_p"] != 0.05 || p["seed"] != float64(benchSeed) {
|
||||
t.Errorf("requête %d : échantillonnage du preset ou seed absents : %v", i+1, p)
|
||||
}
|
||||
if _, ok := p["ignore_eos"]; ok {
|
||||
t.Errorf("requête %d : ignore_eos envoyé — génération hors distribution", i+1)
|
||||
}
|
||||
if kw, _ := p["chat_template_kwargs"].(map[string]any); kw["enable_thinking"] != false {
|
||||
t.Errorf("requête %d : REASONING=off non transmis au gabarit : %v", i+1, p["chat_template_kwargs"])
|
||||
}
|
||||
}
|
||||
// Chaque tour apporte du code JAMAIS vu : aucun morceau ne se répète.
|
||||
seen := map[string]bool{}
|
||||
for _, m := range f.calls[5]["messages"].([]any) {
|
||||
c := m.(map[string]any)["content"].(string)
|
||||
if c != "réponse" && seen[c] {
|
||||
t.Error("un morceau de corpus est renvoyé deux fois")
|
||||
}
|
||||
seen[c] = true
|
||||
}
|
||||
if len(phases) != 6 || !strings.HasPrefix(phases[5], "6/6 ") {
|
||||
t.Errorf("progression inattendue : %v", phases)
|
||||
}
|
||||
d := res.Depth
|
||||
if d == nil || d.Skipped != "" || d.Partial != "" || d.Cold == nil || len(d.Turns) != benchTurns {
|
||||
t.Fatalf("phase en profondeur incomplète : %+v", d)
|
||||
}
|
||||
if want, _ := benchDepthFor(16384, false, 2*512+benchWarmGen); d.Target != want || d.Cold.New < want*9/10 {
|
||||
t.Errorf("profondeur : visée %d (attendu %d), froid %d jetons", d.Target, want, d.Cold.New)
|
||||
}
|
||||
if d.Reuse < 0.99 || d.ReuseNote != "" {
|
||||
t.Errorf("reprise du cache : %v %q, attendu complète", d.Reuse, d.ReuseNote)
|
||||
}
|
||||
if d.DecodePerSec != 50 || d.CachedPerSec != 2000 {
|
||||
t.Errorf("agrégats : decode %v, prefill %v", d.DecodePerSec, d.CachedPerSec)
|
||||
}
|
||||
if d.Draft == nil || d.Draft.N != 30 || d.Draft.Accepted != 21 || res.Draft == nil || res.Draft.N != 10 {
|
||||
t.Errorf("acceptation du brouillon : profondeur %+v, ligne courte %+v", d.Draft, res.Draft)
|
||||
}
|
||||
if res.Engine != "b9999-test" || res.Protocol != benchProtocol || res.Mode != benchModeFull || !benchSavable(res) {
|
||||
t.Errorf("métadonnées : %+v", res)
|
||||
}
|
||||
}
|
||||
|
||||
// Le mode rapide s'arrête à la ligne courte.
|
||||
func TestBenchModeRapide(t *testing.T) {
|
||||
f := &fakeBenchEngine{nCtx: 16384}
|
||||
eng, _ := f.start(t)
|
||||
res, err := benchRun(context.Background(), eng, benchOpts{Mode: benchModeQuick}, testBenchSetup(), testBenchCorpora(), nil)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(f.calls) != 2 || res.Depth != nil || res.PredictedPerSec != 50 {
|
||||
t.Fatalf("%d requêtes, profondeur %+v, decode %v", len(f.calls), res.Depth, res.PredictedPerSec)
|
||||
}
|
||||
}
|
||||
|
||||
// Une réponse non-200 ou sans timings n'est jamais une mesure : erreur, donc
|
||||
// rien d'enregistré (runBench n'enregistre qu'un résultat sans erreur).
|
||||
func TestBenchEchecsNonEnregistres(t *testing.T) {
|
||||
for _, c := range []struct {
|
||||
name string
|
||||
f *fakeBenchEngine
|
||||
want string
|
||||
}{
|
||||
{"500 sur la ligne courte", &fakeBenchEngine{nCtx: 16384, failAt: 2}, "ligne courte"},
|
||||
{"500 sur un tour", &fakeBenchEngine{nCtx: 16384, failAt: 5}, "tour 2"},
|
||||
{"sans timings", &fakeBenchEngine{nCtx: 16384, noTiming: true}, errBenchNoTimings.Error()},
|
||||
{"moteur occupé", &fakeBenchEngine{nCtx: 16384, busy: true}, "déjà une requête"},
|
||||
} {
|
||||
t.Run(c.name, func(t *testing.T) {
|
||||
eng, _ := c.f.start(t)
|
||||
res, err := benchRun(context.Background(), eng, benchOpts{Mode: benchModeFull}, testBenchSetup(), testBenchCorpora(), nil)
|
||||
if err == nil || res != nil || !strings.Contains(err.Error(), c.want) {
|
||||
t.Fatalf("res=%v err=%v, attendu une erreur contenant %q", res, err, c.want)
|
||||
}
|
||||
if c.f.busy && len(c.f.calls) != 0 {
|
||||
t.Error("moteur occupé : aucune requête ne devait partir")
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// Contexte trop petit : la profondeur est sautée, la ligne courte reste.
|
||||
func TestBenchProfondeurSautee(t *testing.T) {
|
||||
f := &fakeBenchEngine{nCtx: 8192}
|
||||
eng, _ := f.start(t)
|
||||
res, err := benchRun(context.Background(), eng, benchOpts{Mode: benchModeFull}, testBenchSetup(), testBenchCorpora(), nil)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if res.Depth == nil || res.Depth.Skipped == "" || len(f.calls) != 2 || !benchSavable(res) {
|
||||
t.Fatalf("profondeur %+v, %d requêtes", res.Depth, len(f.calls))
|
||||
}
|
||||
}
|
||||
|
||||
// Un tour qui dépasse son budget rend un résultat PARTIEL, montré mais jamais
|
||||
// enregistré.
|
||||
func TestBenchBudgetPartiel(t *testing.T) {
|
||||
old := benchReqTimeout
|
||||
benchReqTimeout = 200 * time.Millisecond
|
||||
t.Cleanup(func() { benchReqTimeout = old })
|
||||
f := &fakeBenchEngine{nCtx: 16384, slowHot: 2 * time.Second}
|
||||
eng, _ := f.start(t)
|
||||
res, err := benchRun(context.Background(), eng, benchOpts{Mode: benchModeFull}, testBenchSetup(), testBenchCorpora(), nil)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if res.Depth == nil || !strings.Contains(res.Depth.Partial, "tour 1") || benchSavable(res) {
|
||||
t.Fatalf("attendu un résultat partiel non enregistrable : %+v", res.Depth)
|
||||
}
|
||||
}
|
||||
|
||||
// Annulé : erreur, rien d'enregistré.
|
||||
func TestBenchAnnule(t *testing.T) {
|
||||
f := &fakeBenchEngine{nCtx: 16384, slowHot: 5 * time.Second}
|
||||
eng, _ := f.start(t)
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
go func() { time.Sleep(300 * time.Millisecond); cancel() }()
|
||||
res, err := benchRun(ctx, eng, benchOpts{Mode: benchModeFull}, testBenchSetup(), testBenchCorpora(), nil)
|
||||
if err == nil || res != nil {
|
||||
t.Fatalf("annulation : res=%v err=%v", res, err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBenchDepthFor(t *testing.T) {
|
||||
warm := 2*512 + benchWarmGen
|
||||
for _, c := range []struct {
|
||||
ctx int
|
||||
cpu bool
|
||||
want int
|
||||
skipped bool
|
||||
comments string
|
||||
}{
|
||||
{131072, false, benchDepthMax, false, "grand contexte : plafond"},
|
||||
{131072, true, benchDepthCPU, false, "poids sur CPU : plafond réduit"},
|
||||
{65536, false, 32768, false, "moitié du contexte"},
|
||||
{16384, false, 16384 - benchTurns*(benchAppendTokens+benchGenTokens) - benchGenTokens - warm - benchMargin, false, "les tours doivent tenir"},
|
||||
{8192, false, 0, true, "trop petit"},
|
||||
{0, false, 0, true, "inconnu"},
|
||||
} {
|
||||
got, why := benchDepthFor(c.ctx, c.cpu, warm)
|
||||
if got != c.want || (why != "") != c.skipped {
|
||||
t.Errorf("%s : benchDepthFor(%d) = %d %q", c.comments, c.ctx, got, why)
|
||||
}
|
||||
if got > 0 && got+benchTurns*(benchAppendTokens+benchGenTokens)+benchGenTokens > c.ctx {
|
||||
t.Errorf("%s : %d + tours déborde %d", c.comments, got, c.ctx)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestBenchKVLabel(t *testing.T) {
|
||||
for _, c := range []struct {
|
||||
k, v, label string
|
||||
quant bool
|
||||
}{
|
||||
{"", "", "f16", false},
|
||||
{"f16", "", "f16", false},
|
||||
{"bf16", "bf16", "bf16", false},
|
||||
{"q8_0", "q8_0", "q8_0", true},
|
||||
{"q8_0", "", "q8_0/f16", true},
|
||||
{"", "q4_0", "f16/q4_0", true},
|
||||
} {
|
||||
label, quant := benchKVLabel(c.k, c.v)
|
||||
if label != c.label || quant != c.quant {
|
||||
t.Errorf("benchKVLabel(%q,%q) = %q,%v ; attendu %q,%v", c.k, c.v, label, quant, c.label, c.quant)
|
||||
}
|
||||
}
|
||||
// Le -ctk d'EXTRA_ARGS l'emporte, comme au lancement.
|
||||
s := benchSetupFrom(map[string]string{"KV_TYPE": "f16", "EXTRA_ARGS": "-ctk q8_0 -ctv q8_0 --n-cpu-moe 20"}, nil)
|
||||
if s.kv != "q8_0" || !s.kvQuant || !s.cpuPlaced {
|
||||
t.Errorf("réglages tirés du preset : %+v", s)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBenchPayload(t *testing.T) {
|
||||
for _, c := range []struct {
|
||||
name string
|
||||
cfg map[string]string
|
||||
want map[string]any
|
||||
none []string
|
||||
}{
|
||||
{"échantillonnage du preset", map[string]string{"TEMP": "0.6", "TOP_P": "0.95", "TOP_K": "20"},
|
||||
map[string]any{"temperature": 0.6, "top_p": 0.95, "top_k": 20, "seed": benchSeed}, []string{"chat_template_kwargs", "reasoning_effort"}},
|
||||
{"sans réglage", map[string]string{},
|
||||
map[string]any{"temperature": 0.7, "seed": benchSeed}, []string{"top_k", "chat_template_kwargs"}},
|
||||
{"effort explicite", map[string]string{"REASONING_EFFORT": "high"},
|
||||
map[string]any{"reasoning_effort": "high"}, nil},
|
||||
} {
|
||||
t.Run(c.name, func(t *testing.T) {
|
||||
testHome(t) // effortResolve lit le modèle de la configuration
|
||||
p := benchSetupFrom(c.cfg, nil).payload([]Message{{Role: "user", Content: "x"}}, 10, true)
|
||||
for k, v := range c.want {
|
||||
if fmt.Sprint(p[k]) != fmt.Sprint(v) {
|
||||
t.Errorf("%s = %v, attendu %v", k, p[k], v)
|
||||
}
|
||||
}
|
||||
for _, k := range append(c.none, "ignore_eos") {
|
||||
if _, ok := p[k]; ok {
|
||||
t.Errorf("%s ne devait pas être envoyé : %v", k, p[k])
|
||||
}
|
||||
}
|
||||
if p["cache_prompt"] != true || p["stream"] != false {
|
||||
t.Errorf("cache_prompt/stream : %v", p)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestBenchSlice(t *testing.T) {
|
||||
text := strings.Repeat("ééé ligne\n", 100)
|
||||
off := 0
|
||||
for off < len(text) {
|
||||
chunk, next := benchSlice(text, off, 37)
|
||||
if chunk == "" || !utf8.ValidString(chunk) {
|
||||
t.Fatalf("morceau invalide à %d : %q", off, chunk)
|
||||
}
|
||||
if next < len(text) && !strings.HasSuffix(chunk, "\n") {
|
||||
t.Errorf("morceau coupé hors fin de ligne : %q", chunk)
|
||||
}
|
||||
off = next
|
||||
}
|
||||
if c, _ := benchSlice(text, len(text), 10); c != "" {
|
||||
t.Error("corpus épuisé : morceau vide attendu")
|
||||
}
|
||||
if got := benchLineStart("ab\ncd", 1); got != 3 {
|
||||
t.Errorf("benchLineStart = %d", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBenchReuse(t *testing.T) {
|
||||
full := []benchTurn{{Cached: 1000, Expected: 1004}, {Cached: 3000, Expected: 3000}}
|
||||
if r, note := benchReuse(full, false); r < 0.99 || note != "" {
|
||||
t.Errorf("reprise complète : %v %q", r, note)
|
||||
}
|
||||
part := []benchTurn{{Cached: 512, Expected: 3000}}
|
||||
if _, note := benchReuse(part, true); !strings.Contains(note, "hybride") {
|
||||
t.Errorf("hybride : %q", note)
|
||||
}
|
||||
if _, note := benchReuse(part, false); !strings.Contains(note, "gabarit") {
|
||||
t.Errorf("non hybride : %q", note)
|
||||
}
|
||||
if r, note := benchReuse([]benchTurn{{Cached: -1, Expected: 3000}}, false); r != -1 || note == "" {
|
||||
t.Errorf("inconnu : %v %q", r, note)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBenchThrashHint(t *testing.T) {
|
||||
if h := benchThrashHint(100<<20, 256, 0); h != "" {
|
||||
t.Errorf("100 Mo / 256 jetons : pas d'indication attendue, %q", h)
|
||||
}
|
||||
if h := benchThrashHint(3<<30, 768, 60<<30); !strings.Contains(h, "thrash") {
|
||||
t.Errorf("1 Go / 256 jetons : indication attendue, %q", h)
|
||||
}
|
||||
}
|
||||
|
||||
// Une mesure enregistrée ne s'affiche que pour la configuration et les cartes
|
||||
// sur lesquelles elle a été prise ; une mesure d'avant l'empreinte retombe sur
|
||||
// le nom du modèle.
|
||||
func TestBenchEmpreinte(t *testing.T) {
|
||||
cfg := map[string]string{"MODEL": "/models/Q.gguf", "CTX": "32768", "CUDA_VISIBLE_DEVICES": "0,1"}
|
||||
res := &benchResult{PromptPerSecond: 1, Depth: &benchDepth{Hint: "possible thrash"}}
|
||||
sb := newSavedBench(res, cfg)
|
||||
if sb.Fingerprint == "" || sb.Result.Depth.Hint != "" || res.Depth.Hint == "" {
|
||||
t.Fatalf("enregistrement : %+v (l'indication ne s'enregistre pas, le résultat montré la garde)", sb)
|
||||
}
|
||||
preset := map[string]string{"MODEL": "/models/Q.gguf", "CTX": "32768"}
|
||||
if !benchMatchesPreset(sb, preset, cfg) {
|
||||
t.Error("même configuration, mêmes cartes : la pastille doit s'afficher")
|
||||
}
|
||||
if benchMatchesPreset(sb, map[string]string{"MODEL": "/models/Q.gguf", "CTX": "65536"}, cfg) {
|
||||
t.Error("contexte différent : pastille masquée attendue")
|
||||
}
|
||||
if benchMatchesPreset(sb, preset, map[string]string{"CUDA_VISIBLE_DEVICES": "1,0"}) {
|
||||
t.Error("cartes dans un autre ordre : pastille masquée attendue")
|
||||
}
|
||||
own := map[string]string{"MODEL": "/models/Q.gguf", "CTX": "32768", "CUDA_VISIBLE_DEVICES": "0,1"}
|
||||
if !benchMatchesPreset(sb, own, map[string]string{"CUDA_VISIBLE_DEVICES": "0"}) {
|
||||
t.Error("preset qui impose ses cartes : ce sont elles qui comptent")
|
||||
}
|
||||
legacy := savedBench{Model: "Q.gguf"}
|
||||
if !benchMatchesPreset(legacy, preset, cfg) {
|
||||
t.Error("mesure d'avant l'empreinte : repli sur le nom du modèle")
|
||||
}
|
||||
}
|
||||
|
||||
// Le lancement par l'interface : POST seulement, 409 pendant une génération,
|
||||
// puis un seul bench à la fois, qui tient le verrou du chat et s'annule.
|
||||
func TestBenchHandlers(t *testing.T) {
|
||||
testHome(t)
|
||||
f := &fakeBenchEngine{nCtx: 16384}
|
||||
_, srv := f.start(t)
|
||||
u, _ := url.Parse(srv.URL)
|
||||
if err := WriteConfig(map[string]string{"PORT": u.Port()}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
started := make(chan struct{})
|
||||
old := benchRunner
|
||||
benchRunner = func(ctx context.Context, o benchOpts, p benchProgress) (*benchResult, error) {
|
||||
p("mesure", 1, 2)
|
||||
close(started)
|
||||
<-ctx.Done()
|
||||
return nil, ctx.Err()
|
||||
}
|
||||
t.Cleanup(func() { benchRunner = old })
|
||||
|
||||
do := func(method, path, body string) *httptest.ResponseRecorder {
|
||||
rec := httptest.NewRecorder()
|
||||
req := httptest.NewRequest(method, path, strings.NewReader(body))
|
||||
switch path {
|
||||
case "/api/bench":
|
||||
handleBench(rec, req)
|
||||
case "/api/bench/cancel":
|
||||
handleBenchCancel(rec, req)
|
||||
default:
|
||||
handleBenchStatus(rec, req)
|
||||
}
|
||||
return rec
|
||||
}
|
||||
if rec := do(http.MethodGet, "/api/bench", ""); rec.Code != http.StatusMethodNotAllowed || rec.Header().Get("Allow") != http.MethodPost {
|
||||
t.Fatalf("GET : %d, Allow=%q", rec.Code, rec.Header().Get("Allow"))
|
||||
}
|
||||
|
||||
conv.mu.Lock()
|
||||
conv.Generating = true
|
||||
conv.mu.Unlock()
|
||||
rec := do(http.MethodPost, "/api/bench", `{"mode":"full"}`)
|
||||
conv.mu.Lock()
|
||||
conv.Generating = false
|
||||
conv.mu.Unlock()
|
||||
if rec.Code != http.StatusConflict {
|
||||
t.Fatalf("pendant une génération : %d, attendu 409", rec.Code)
|
||||
}
|
||||
|
||||
if rec := do(http.MethodPost, "/api/bench", `{"mode":"full"}`); rec.Code != http.StatusAccepted {
|
||||
t.Fatalf("lancement : %d %s", rec.Code, rec.Body)
|
||||
}
|
||||
<-started
|
||||
if rec := do(http.MethodPost, "/api/bench", `{}`); rec.Code != http.StatusConflict {
|
||||
t.Errorf("second bench : %d, attendu 409", rec.Code)
|
||||
}
|
||||
if err := conv.StartTurn("bonjour", nil, Caps{}, 0); err != ErrBusy {
|
||||
t.Errorf("tour de chat pendant le bench : %v, attendu ErrBusy", err)
|
||||
}
|
||||
if err := conv.busyReason(); err != errBenchBusy {
|
||||
t.Errorf("motif du refus : %v", err)
|
||||
}
|
||||
if queued, err := conv.EnqueueOrStart("cid-bench", "x", nil, Caps{}, 0); queued || err == nil {
|
||||
t.Errorf("message pendant le bench : mis en file=%v err=%v, attendu un refus", queued, err)
|
||||
}
|
||||
var st map[string]any
|
||||
json.Unmarshal(do(http.MethodGet, "/api/bench/status", "").Body.Bytes(), &st)
|
||||
if st["running"] != true || st["phase"] != "mesure" || st["mode"] != benchModeFull {
|
||||
t.Errorf("statut en cours : %v", st)
|
||||
}
|
||||
if rec := do(http.MethodPost, "/api/bench/cancel", ""); rec.Code != 200 {
|
||||
t.Fatalf("annulation : %d", rec.Code)
|
||||
}
|
||||
deadline := time.Now().Add(3 * time.Second)
|
||||
for benchRunning() && time.Now().Before(deadline) {
|
||||
time.Sleep(10 * time.Millisecond)
|
||||
}
|
||||
json.Unmarshal(do(http.MethodGet, "/api/bench/status", "").Body.Bytes(), &st)
|
||||
if st["running"] != false || st["canceled"] != true {
|
||||
t.Errorf("statut après annulation : %v", st)
|
||||
}
|
||||
if conv.isGenerating() {
|
||||
t.Error("verrou de génération non rendu après le bench")
|
||||
}
|
||||
}
|
||||
|
||||
// Les clients /v1 reçoivent un 503 à retenter pendant un benchmark.
|
||||
func TestBenchRefuseV1(t *testing.T) {
|
||||
benchJob.mu.Lock()
|
||||
benchJob.running = true
|
||||
benchJob.mu.Unlock()
|
||||
t.Cleanup(func() {
|
||||
benchJob.mu.Lock()
|
||||
benchJob.running = false
|
||||
benchJob.mu.Unlock()
|
||||
})
|
||||
rec := httptest.NewRecorder()
|
||||
oaiHandler().ServeHTTP(rec, httptest.NewRequest(http.MethodPost, "/v1/chat/completions", strings.NewReader("{}")))
|
||||
if rec.Code != http.StatusServiceUnavailable || rec.Header().Get("Retry-After") == "" {
|
||||
t.Fatalf("pendant le bench : %d, Retry-After=%q", rec.Code, rec.Header().Get("Retry-After"))
|
||||
}
|
||||
}
|
||||
@@ -119,6 +119,14 @@ func oaiHandler() http.Handler {
|
||||
// Complétion d'un client externe : elle prend le slot sous le nez de
|
||||
// la conversation. Notée pour nommer la perte de cache qui suit.
|
||||
if r.Method == http.MethodPost && strings.HasPrefix(p, "/v1/") {
|
||||
// Pendant un benchmark, elle fausserait la mesure et attendrait des
|
||||
// minutes derrière lui : refus explicite, à retenter.
|
||||
if benchRunning() {
|
||||
w.Header().Set("Retry-After", "30")
|
||||
sendOAIError(w, http.StatusServiceUnavailable,
|
||||
"benchmark en cours sur ce moteur — réessaie dans un moment", "server_error", "benchmark_running")
|
||||
return
|
||||
}
|
||||
perfNoteForeign()
|
||||
}
|
||||
lp.ServeHTTP(w, r)
|
||||
|
||||
@@ -101,6 +101,7 @@ type perfWire struct {
|
||||
PromptMs perfNum `json:"prompt_ms"`
|
||||
PromptPerSecond perfNum `json:"prompt_per_second"`
|
||||
PredictedN perfNum `json:"predicted_n"`
|
||||
PredictedMs perfNum `json:"predicted_ms"`
|
||||
PredictedPerSec perfNum `json:"predicted_per_second"`
|
||||
CacheN perfNum `json:"cache_n"`
|
||||
DraftN perfNum `json:"draft_n"`
|
||||
|
||||
@@ -135,7 +135,7 @@ Moteur (loki-engine) :
|
||||
enable | disable auto-démarrage au boot
|
||||
edit éditer la configuration dans $EDITOR
|
||||
switch [N] activer un preset de presets/ (interactif ou par numéro)
|
||||
test | bench [N] vérifier que l'IA répond / mesurer prefill + decode tok/s
|
||||
test | bench [N] [--full] vérifier que l'IA répond / mesurer prefill + decode (--full : en profondeur)
|
||||
vram utilisation GPU/VRAM (nvidia-smi)
|
||||
gpu [index…] liste les GPU / choisit le(s)quel(s) utiliser (gpu all = tous)
|
||||
set-api-key [clé] protéger llama-server (clé Bearer); vide = générer, "" = retirer
|
||||
|
||||
@@ -102,3 +102,27 @@ func modelTotalSize(path string) int64 {
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// engineIOSample : octets lus du disque (read_bytes) et mémoire résidente du
|
||||
// moteur, pour l'indication « possible thrash » du benchmark. ok=false si le
|
||||
// PID ou /proc est illisible : le bench s'en passe sans rien dire.
|
||||
func engineIOSample() (read, rss int64, ok bool) {
|
||||
pid := readServicePID()
|
||||
if pid <= 0 {
|
||||
return 0, 0, false
|
||||
}
|
||||
b, err := os.ReadFile("/proc/" + strconv.Itoa(pid) + "/io")
|
||||
if err != nil {
|
||||
return 0, 0, false
|
||||
}
|
||||
for _, line := range strings.Split(string(b), "\n") {
|
||||
if v, found := strings.CutPrefix(line, "read_bytes:"); found {
|
||||
n, err := strconv.ParseInt(strings.TrimSpace(v), 10, 64)
|
||||
if err != nil {
|
||||
return 0, 0, false
|
||||
}
|
||||
return n, procVmRSS(pid), true
|
||||
}
|
||||
}
|
||||
return 0, 0, false
|
||||
}
|
||||
@@ -5,3 +5,6 @@ package loki
|
||||
// Hors Linux, pas de /proc/<pid>/io : le pourcentage de chargement n'est pas
|
||||
// mesurable, l'UI garde « chargement… » sans chiffre.
|
||||
func engineLoadPct() int { return -1 }
|
||||
|
||||
// engineIOSample : pas de /proc/<pid>/io, pas d'indication « possible thrash ».
|
||||
func engineIOSample() (read, rss int64, ok bool) { return 0, 0, false }
|
||||
+96
-36
@@ -2471,7 +2471,9 @@ html[data-files="1"] #files-btn{color:var(--accent)}
|
||||
</div>
|
||||
<div id="bench-body" style="padding:18px 20px;display:flex;flex-direction:column;gap:12px"></div>
|
||||
<div style="padding:12px 16px;border-top:1px solid var(--border);display:flex;justify-content:flex-end;gap:6px">
|
||||
<button id="bench-cancel" onclick="cancelBenchUI()" style="display:none">annuler</button>
|
||||
<button onclick="closeBenchModal()">fermer</button>
|
||||
<button id="bench-full" onclick="runBenchUI('full')" title="prefill à froid en profondeur puis 3 tours qui reprennent le cache — 1 à 5 minutes">bench complet</button>
|
||||
<button id="bench-rerun" onclick="runBenchUI()" style="border-color:var(--accent);color:var(--accent)">relancer</button>
|
||||
</div>
|
||||
</div>
|
||||
@@ -4435,8 +4437,11 @@ async function loadPresets(){
|
||||
}
|
||||
if(x.bench){
|
||||
const bt=document.createElement('span'); bt.className='btag';
|
||||
bt.title='prefill / decode — dernier bench de ce preset';
|
||||
bt.textContent=x.bench.prefill.toFixed(0)+'-'+x.bench.decode.toFixed(0)+' t/s';
|
||||
let tip='prefill / decode — dernier bench de ce preset';
|
||||
if(x.bench.depth_decode) tip+='\ndecode à '+x.bench.depth+' jetons : '+x.bench.depth_decode.toFixed(1)+' t/s';
|
||||
if(x.bench.kv) tip+='\ncache KV '+x.bench.kv+(x.bench.kv_quantized?' — quantifié (zone grise, pas une référence sans perte)':'');
|
||||
bt.title=tip;
|
||||
bt.textContent=x.bench.prefill.toFixed(0)+'-'+x.bench.decode.toFixed(0)+' t/s'+(x.bench.kv_quantized?' · KV '+x.bench.kv:'');
|
||||
meta.appendChild(bt);
|
||||
}
|
||||
info.appendChild(nm);
|
||||
@@ -4963,46 +4968,101 @@ async function loadAll(){
|
||||
async function act(a){ toast(a+'…'); await jpost('/api/'+a); setTimeout(loadAll,1500); }
|
||||
function openBenchModal(){ showModal('bench-modal'); }
|
||||
function closeBenchModal(){ hideModal('bench-modal'); }
|
||||
async function runBenchUI(){
|
||||
const btn = document.getElementById('btn-bench');
|
||||
const rerun = document.getElementById('bench-rerun');
|
||||
const body = document.getElementById('bench-body');
|
||||
openBenchModal();
|
||||
btn.disabled = true; btn.textContent = '⏳ bench…';
|
||||
rerun.disabled = true;
|
||||
body.innerHTML =
|
||||
'<div style="text-align:center;padding:20px 0">' +
|
||||
// Le bench tourne en tâche de fond côté serveur (llm_bench_job.go) : on le lance,
|
||||
// puis on suit sa progression. Fermer la fenêtre ne l'arrête pas ; la rouvrir
|
||||
// (bouton bench) reprend le suivi au lieu d'en lancer un second.
|
||||
let benchPoll=null;
|
||||
function benchButtons(busy){
|
||||
const btn=document.getElementById('btn-bench');
|
||||
btn.disabled=busy; btn.textContent=busy?'⏳ bench…':'bench';
|
||||
for(const id of ['bench-rerun','bench-full']){ const b=document.getElementById(id); if(b) b.disabled=busy; }
|
||||
const c=document.getElementById('bench-cancel'); if(c) c.style.display=busy?'':'none';
|
||||
}
|
||||
function benchSpinner(text){
|
||||
return '<div style="text-align:center;padding:20px 0">' +
|
||||
'<div style="font-size:24px;animation:spin 1s linear infinite;display:inline-block">⏳</div>' +
|
||||
'<div class="muted" style="margin-top:8px">prompt 2000 tok + 300 decode<br>~10 secondes…</div>' +
|
||||
'</div>';
|
||||
'<div class="muted" style="margin-top:8px">'+text+'</div></div>';
|
||||
}
|
||||
function benchErr(msg){ return '<div style="color:var(--err);text-align:center">erreur : '+escHtml(msg||'?')+'</div>'; }
|
||||
async function runBenchUI(mode){
|
||||
mode = mode==='full' ? 'full' : 'quick';
|
||||
const body=document.getElementById('bench-body');
|
||||
openBenchModal();
|
||||
benchButtons(true);
|
||||
body.innerHTML=benchSpinner('démarrage…');
|
||||
try{
|
||||
const r = await jget('/api/bench');
|
||||
if(!r.ok){
|
||||
body.innerHTML = '<div style="color:var(--err);text-align:center">erreur: '+r.error+'</div>';
|
||||
const st=await jget('/api/bench/status');
|
||||
if(st.running){ benchWatch(); return; }
|
||||
}catch(e){}
|
||||
let r;
|
||||
try{ r=await jpost('/api/bench',{mode:mode}); }catch(e){ r={ok:false,error:e.message}; }
|
||||
if(!r.ok){ body.innerHTML=benchErr(r.error); benchButtons(false); return; }
|
||||
benchWatch();
|
||||
}
|
||||
async function cancelBenchUI(){ try{ await jpost('/api/bench/cancel',{}); }catch(e){} }
|
||||
function benchWatch(){
|
||||
clearTimeout(benchPoll);
|
||||
const body=document.getElementById('bench-body');
|
||||
const tick=async()=>{
|
||||
let st;
|
||||
try{ st=await jget('/api/bench/status'); }
|
||||
catch(e){ benchPoll=setTimeout(tick,2000); return; }
|
||||
if(st.running){
|
||||
const step=st.steps?' ('+st.step+'/'+st.steps+')':'';
|
||||
const hint=st.mode==='full'?'<br>bench complet : 1 à 5 minutes, chat et tâches en pause':'';
|
||||
body.innerHTML=benchSpinner(escHtml(st.phase||'…')+step+' · '+Math.round(st.elapsed_sec||0)+' s'+hint);
|
||||
benchPoll=setTimeout(tick,1000);
|
||||
return;
|
||||
}
|
||||
const x = r.result;
|
||||
body.innerHTML =
|
||||
'<div style="display:grid;grid-template-columns:1fr 1fr;gap:16px;text-align:center">' +
|
||||
'<div style="padding:14px;background:var(--panel);border:1px solid var(--border);border-radius:8px">' +
|
||||
'<div class="muted" style="font-size:11px;text-transform:uppercase;letter-spacing:.1em">Prefill</div>' +
|
||||
'<div style="font-size:26px;color:var(--accent);font-weight:600;margin:6px 0">'+x.prompt_per_second.toFixed(0)+'</div>' +
|
||||
'<div class="muted">tok/s</div>' +
|
||||
'<div class="muted" style="font-size:11px;margin-top:8px">'+x.prompt_n+' tok · '+(x.prompt_ms/1000).toFixed(2)+'s</div>' +
|
||||
'</div>' +
|
||||
'<div style="padding:14px;background:var(--panel);border:1px solid var(--border);border-radius:8px">' +
|
||||
'<div class="muted" style="font-size:11px;text-transform:uppercase;letter-spacing:.1em">Decode</div>' +
|
||||
'<div style="font-size:26px;color:var(--ok);font-weight:600;margin:6px 0">'+x.predicted_per_second.toFixed(1)+'</div>' +
|
||||
'<div class="muted">tok/s</div>' +
|
||||
'<div class="muted" style="font-size:11px;margin-top:8px">'+x.predicted_n+' tok · '+(x.predicted_ms/1000).toFixed(2)+'s</div>' +
|
||||
'</div>' +
|
||||
'</div>' +
|
||||
'<div class="muted" style="text-align:center;font-size:11px">total '+x.elapsed_sec.toFixed(2)+'s</div>';
|
||||
} finally {
|
||||
btn.disabled = false; btn.textContent = 'bench';
|
||||
rerun.disabled = false;
|
||||
benchButtons(false);
|
||||
if(st.error) body.innerHTML=benchErr(st.error);
|
||||
else if(st.result) body.innerHTML=benchResultHTML(st.result, st.saved);
|
||||
else body.innerHTML='';
|
||||
loadPresets();
|
||||
};
|
||||
tick();
|
||||
}
|
||||
function benchDraftTxt(d){ return d&&d.n ? Math.round(d.accepted*100/d.n)+' % ('+d.accepted+'/'+d.n+')' : 'n/a'; }
|
||||
function benchCard(label, color, val, unit, sub){
|
||||
return '<div style="padding:14px;background:var(--panel);border:1px solid var(--border);border-radius:8px">' +
|
||||
'<div class="muted" style="font-size:11px;text-transform:uppercase;letter-spacing:.1em">'+label+'</div>' +
|
||||
'<div style="font-size:26px;color:'+color+';font-weight:600;margin:6px 0">'+val+'</div>' +
|
||||
'<div class="muted">'+unit+'</div>' +
|
||||
'<div class="muted" style="font-size:11px;margin-top:8px">'+sub+'</div></div>';
|
||||
}
|
||||
function benchResultHTML(x, saved){
|
||||
let h='<div style="display:grid;grid-template-columns:1fr 1fr;gap:16px;text-align:center">' +
|
||||
benchCard('Prefill','var(--accent)',x.prompt_per_second.toFixed(0),'tok/s',x.prompt_n+' tok · '+(x.prompt_ms/1000).toFixed(2)+'s') +
|
||||
benchCard('Decode','var(--ok)',x.predicted_per_second.toFixed(1),'tok/s',x.predicted_n+' tok · '+(x.predicted_ms/1000).toFixed(2)+'s') +
|
||||
'</div>';
|
||||
const rows=[];
|
||||
if(x.draft) rows.push(['brouillon accepté (prose)', benchDraftTxt(x.draft)]);
|
||||
const d=x.depth;
|
||||
const notes=[];
|
||||
if(d){
|
||||
if(d.skipped) notes.push('profondeur sautée : '+d.skipped);
|
||||
if(d.cold) rows.push(['prefill à froid · '+d.cold.new+' tok', d.cold.prompt_per_second.toFixed(0)+' tok/s']);
|
||||
if(d.turns&&d.turns.length){
|
||||
rows.push(['prefill des tours suivants (cache)', d.cached_per_second.toFixed(0)+' tok/s']);
|
||||
rows.push(['decode en profondeur', d.decode_per_second.toFixed(1)+' tok/s']);
|
||||
rows.push(['reprise du cache', d.reuse>=0 ? Math.round(d.reuse*100)+' %' : 'n/a']);
|
||||
rows.push(['brouillon accepté (code)', benchDraftTxt(d.draft)]);
|
||||
}
|
||||
for(const n of [d.reuse_note, d.partial, d.hint]) if(n) notes.push(n);
|
||||
}
|
||||
if(rows.length){
|
||||
h+='<table style="width:100%;font-size:12px;border-collapse:collapse">';
|
||||
for(const r of rows) h+='<tr><td class="muted" style="padding:3px 0">'+escHtml(r[0])+'</td><td style="text-align:right">'+escHtml(r[1])+'</td></tr>';
|
||||
h+='</table>';
|
||||
}
|
||||
for(const n of notes) h+='<div style="font-size:11px;color:var(--warn)">'+escHtml(n)+'</div>';
|
||||
let foot='KV '+(x.kv||'?')+(x.kv_quantized?' (quantifié — zone grise)':'');
|
||||
if(x.engine) foot+=' · moteur '+x.engine;
|
||||
foot+=' · total '+x.elapsed_sec.toFixed(1)+'s';
|
||||
if(saved===false) foot+=' · résultat partiel, non enregistré';
|
||||
h+='<div class="muted" style="text-align:center;font-size:11px">'+escHtml(foot)+'</div>';
|
||||
if(d&&d.cold) h+='<div class="muted" style="text-align:center;font-size:11px">Le slot est effacé après le bench quand le moteur le permet : sinon, le prochain message reprend la conversation depuis le cache RAM (CACHE_RAM) ou la recalcule.</div>';
|
||||
return h;
|
||||
}
|
||||
async function switchTo(n,name,id){
|
||||
if(!await askConfirm('Basculer vers « '+name+' » et redémarrer le service ?', {title:'Changer de preset', okText:'Basculer'})) return;
|
||||
|
||||
@@ -245,7 +245,9 @@ document.documentElement.setAttribute('data-side',localStorage.getItem('loki-sid
|
||||
</div>
|
||||
<div id="bench-body" style="padding:18px 20px;display:flex;flex-direction:column;gap:12px"></div>
|
||||
<div style="padding:12px 16px;border-top:1px solid var(--border);display:flex;justify-content:flex-end;gap:6px">
|
||||
<button id="bench-cancel" onclick="cancelBenchUI()" style="display:none">annuler</button>
|
||||
<button onclick="closeBenchModal()">fermer</button>
|
||||
<button id="bench-full" onclick="runBenchUI('full')" title="prefill à froid en profondeur puis 3 tours qui reprennent le cache — 1 à 5 minutes">bench complet</button>
|
||||
<button id="bench-rerun" onclick="runBenchUI()" style="border-color:var(--accent);color:var(--accent)">relancer</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
@@ -93,8 +93,11 @@ async function loadPresets(){
|
||||
}
|
||||
if(x.bench){
|
||||
const bt=document.createElement('span'); bt.className='btag';
|
||||
bt.title='prefill / decode — dernier bench de ce preset';
|
||||
bt.textContent=x.bench.prefill.toFixed(0)+'-'+x.bench.decode.toFixed(0)+' t/s';
|
||||
let tip='prefill / decode — dernier bench de ce preset';
|
||||
if(x.bench.depth_decode) tip+='\ndecode à '+x.bench.depth+' jetons : '+x.bench.depth_decode.toFixed(1)+' t/s';
|
||||
if(x.bench.kv) tip+='\ncache KV '+x.bench.kv+(x.bench.kv_quantized?' — quantifié (zone grise, pas une référence sans perte)':'');
|
||||
bt.title=tip;
|
||||
bt.textContent=x.bench.prefill.toFixed(0)+'-'+x.bench.decode.toFixed(0)+' t/s'+(x.bench.kv_quantized?' · KV '+x.bench.kv:'');
|
||||
meta.appendChild(bt);
|
||||
}
|
||||
info.appendChild(nm);
|
||||
|
||||
@@ -1,45 +1,100 @@
|
||||
function openBenchModal(){ showModal('bench-modal'); }
|
||||
function closeBenchModal(){ hideModal('bench-modal'); }
|
||||
async function runBenchUI(){
|
||||
const btn = document.getElementById('btn-bench');
|
||||
const rerun = document.getElementById('bench-rerun');
|
||||
const body = document.getElementById('bench-body');
|
||||
openBenchModal();
|
||||
btn.disabled = true; btn.textContent = '⏳ bench…';
|
||||
rerun.disabled = true;
|
||||
body.innerHTML =
|
||||
'<div style="text-align:center;padding:20px 0">' +
|
||||
// Le bench tourne en tâche de fond côté serveur (llm_bench_job.go) : on le lance,
|
||||
// puis on suit sa progression. Fermer la fenêtre ne l'arrête pas ; la rouvrir
|
||||
// (bouton bench) reprend le suivi au lieu d'en lancer un second.
|
||||
let benchPoll=null;
|
||||
function benchButtons(busy){
|
||||
const btn=document.getElementById('btn-bench');
|
||||
btn.disabled=busy; btn.textContent=busy?'⏳ bench…':'bench';
|
||||
for(const id of ['bench-rerun','bench-full']){ const b=document.getElementById(id); if(b) b.disabled=busy; }
|
||||
const c=document.getElementById('bench-cancel'); if(c) c.style.display=busy?'':'none';
|
||||
}
|
||||
function benchSpinner(text){
|
||||
return '<div style="text-align:center;padding:20px 0">' +
|
||||
'<div style="font-size:24px;animation:spin 1s linear infinite;display:inline-block">⏳</div>' +
|
||||
'<div class="muted" style="margin-top:8px">prompt 2000 tok + 300 decode<br>~10 secondes…</div>' +
|
||||
'</div>';
|
||||
'<div class="muted" style="margin-top:8px">'+text+'</div></div>';
|
||||
}
|
||||
function benchErr(msg){ return '<div style="color:var(--err);text-align:center">erreur : '+escHtml(msg||'?')+'</div>'; }
|
||||
async function runBenchUI(mode){
|
||||
mode = mode==='full' ? 'full' : 'quick';
|
||||
const body=document.getElementById('bench-body');
|
||||
openBenchModal();
|
||||
benchButtons(true);
|
||||
body.innerHTML=benchSpinner('démarrage…');
|
||||
try{
|
||||
const r = await jget('/api/bench');
|
||||
if(!r.ok){
|
||||
body.innerHTML = '<div style="color:var(--err);text-align:center">erreur: '+r.error+'</div>';
|
||||
const st=await jget('/api/bench/status');
|
||||
if(st.running){ benchWatch(); return; }
|
||||
}catch(e){}
|
||||
let r;
|
||||
try{ r=await jpost('/api/bench',{mode:mode}); }catch(e){ r={ok:false,error:e.message}; }
|
||||
if(!r.ok){ body.innerHTML=benchErr(r.error); benchButtons(false); return; }
|
||||
benchWatch();
|
||||
}
|
||||
async function cancelBenchUI(){ try{ await jpost('/api/bench/cancel',{}); }catch(e){} }
|
||||
function benchWatch(){
|
||||
clearTimeout(benchPoll);
|
||||
const body=document.getElementById('bench-body');
|
||||
const tick=async()=>{
|
||||
let st;
|
||||
try{ st=await jget('/api/bench/status'); }
|
||||
catch(e){ benchPoll=setTimeout(tick,2000); return; }
|
||||
if(st.running){
|
||||
const step=st.steps?' ('+st.step+'/'+st.steps+')':'';
|
||||
const hint=st.mode==='full'?'<br>bench complet : 1 à 5 minutes, chat et tâches en pause':'';
|
||||
body.innerHTML=benchSpinner(escHtml(st.phase||'…')+step+' · '+Math.round(st.elapsed_sec||0)+' s'+hint);
|
||||
benchPoll=setTimeout(tick,1000);
|
||||
return;
|
||||
}
|
||||
const x = r.result;
|
||||
body.innerHTML =
|
||||
'<div style="display:grid;grid-template-columns:1fr 1fr;gap:16px;text-align:center">' +
|
||||
'<div style="padding:14px;background:var(--panel);border:1px solid var(--border);border-radius:8px">' +
|
||||
'<div class="muted" style="font-size:11px;text-transform:uppercase;letter-spacing:.1em">Prefill</div>' +
|
||||
'<div style="font-size:26px;color:var(--accent);font-weight:600;margin:6px 0">'+x.prompt_per_second.toFixed(0)+'</div>' +
|
||||
'<div class="muted">tok/s</div>' +
|
||||
'<div class="muted" style="font-size:11px;margin-top:8px">'+x.prompt_n+' tok · '+(x.prompt_ms/1000).toFixed(2)+'s</div>' +
|
||||
'</div>' +
|
||||
'<div style="padding:14px;background:var(--panel);border:1px solid var(--border);border-radius:8px">' +
|
||||
'<div class="muted" style="font-size:11px;text-transform:uppercase;letter-spacing:.1em">Decode</div>' +
|
||||
'<div style="font-size:26px;color:var(--ok);font-weight:600;margin:6px 0">'+x.predicted_per_second.toFixed(1)+'</div>' +
|
||||
'<div class="muted">tok/s</div>' +
|
||||
'<div class="muted" style="font-size:11px;margin-top:8px">'+x.predicted_n+' tok · '+(x.predicted_ms/1000).toFixed(2)+'s</div>' +
|
||||
'</div>' +
|
||||
'</div>' +
|
||||
'<div class="muted" style="text-align:center;font-size:11px">total '+x.elapsed_sec.toFixed(2)+'s</div>';
|
||||
} finally {
|
||||
btn.disabled = false; btn.textContent = 'bench';
|
||||
rerun.disabled = false;
|
||||
benchButtons(false);
|
||||
if(st.error) body.innerHTML=benchErr(st.error);
|
||||
else if(st.result) body.innerHTML=benchResultHTML(st.result, st.saved);
|
||||
else body.innerHTML='';
|
||||
loadPresets();
|
||||
};
|
||||
tick();
|
||||
}
|
||||
function benchDraftTxt(d){ return d&&d.n ? Math.round(d.accepted*100/d.n)+' % ('+d.accepted+'/'+d.n+')' : 'n/a'; }
|
||||
function benchCard(label, color, val, unit, sub){
|
||||
return '<div style="padding:14px;background:var(--panel);border:1px solid var(--border);border-radius:8px">' +
|
||||
'<div class="muted" style="font-size:11px;text-transform:uppercase;letter-spacing:.1em">'+label+'</div>' +
|
||||
'<div style="font-size:26px;color:'+color+';font-weight:600;margin:6px 0">'+val+'</div>' +
|
||||
'<div class="muted">'+unit+'</div>' +
|
||||
'<div class="muted" style="font-size:11px;margin-top:8px">'+sub+'</div></div>';
|
||||
}
|
||||
function benchResultHTML(x, saved){
|
||||
let h='<div style="display:grid;grid-template-columns:1fr 1fr;gap:16px;text-align:center">' +
|
||||
benchCard('Prefill','var(--accent)',x.prompt_per_second.toFixed(0),'tok/s',x.prompt_n+' tok · '+(x.prompt_ms/1000).toFixed(2)+'s') +
|
||||
benchCard('Decode','var(--ok)',x.predicted_per_second.toFixed(1),'tok/s',x.predicted_n+' tok · '+(x.predicted_ms/1000).toFixed(2)+'s') +
|
||||
'</div>';
|
||||
const rows=[];
|
||||
if(x.draft) rows.push(['brouillon accepté (prose)', benchDraftTxt(x.draft)]);
|
||||
const d=x.depth;
|
||||
const notes=[];
|
||||
if(d){
|
||||
if(d.skipped) notes.push('profondeur sautée : '+d.skipped);
|
||||
if(d.cold) rows.push(['prefill à froid · '+d.cold.new+' tok', d.cold.prompt_per_second.toFixed(0)+' tok/s']);
|
||||
if(d.turns&&d.turns.length){
|
||||
rows.push(['prefill des tours suivants (cache)', d.cached_per_second.toFixed(0)+' tok/s']);
|
||||
rows.push(['decode en profondeur', d.decode_per_second.toFixed(1)+' tok/s']);
|
||||
rows.push(['reprise du cache', d.reuse>=0 ? Math.round(d.reuse*100)+' %' : 'n/a']);
|
||||
rows.push(['brouillon accepté (code)', benchDraftTxt(d.draft)]);
|
||||
}
|
||||
for(const n of [d.reuse_note, d.partial, d.hint]) if(n) notes.push(n);
|
||||
}
|
||||
if(rows.length){
|
||||
h+='<table style="width:100%;font-size:12px;border-collapse:collapse">';
|
||||
for(const r of rows) h+='<tr><td class="muted" style="padding:3px 0">'+escHtml(r[0])+'</td><td style="text-align:right">'+escHtml(r[1])+'</td></tr>';
|
||||
h+='</table>';
|
||||
}
|
||||
for(const n of notes) h+='<div style="font-size:11px;color:var(--warn)">'+escHtml(n)+'</div>';
|
||||
let foot='KV '+(x.kv||'?')+(x.kv_quantized?' (quantifié — zone grise)':'');
|
||||
if(x.engine) foot+=' · moteur '+x.engine;
|
||||
foot+=' · total '+x.elapsed_sec.toFixed(1)+'s';
|
||||
if(saved===false) foot+=' · résultat partiel, non enregistré';
|
||||
h+='<div class="muted" style="text-align:center;font-size:11px">'+escHtml(foot)+'</div>';
|
||||
if(d&&d.cold) h+='<div class="muted" style="text-align:center;font-size:11px">Le slot est effacé après le bench quand le moteur le permet : sinon, le prochain message reprend la conversation depuis le cache RAM (CACHE_RAM) ou la recalcule.</div>';
|
||||
return h;
|
||||
}
|
||||
async function switchTo(n,name,id){
|
||||
if(!await askConfirm('Basculer vers « '+name+' » et redémarrer le service ?', {title:'Changer de preset', okText:'Basculer'})) return;
|
||||
|
||||
+14
-35
@@ -408,6 +408,7 @@ func handlePresets(w http.ResponseWriter, r *http.Request) {
|
||||
return
|
||||
}
|
||||
store := loadBenchStore()
|
||||
cur := ReadConfig()
|
||||
out := []map[string]any{}
|
||||
for _, p := range list {
|
||||
item := map[string]any{"id": p.ID, "name": p.Name, "active": p.Active}
|
||||
@@ -430,12 +431,23 @@ func handlePresets(w http.ResponseWriter, r *http.Request) {
|
||||
item["reasoning"] = strings.ToLower(r)
|
||||
}
|
||||
}
|
||||
if sb, ok := store[p.ID]; ok && benchMatchesPreset(sb, parseEnv(content)) {
|
||||
item["bench"] = map[string]any{
|
||||
if sb, ok := store[p.ID]; ok && benchMatchesPreset(sb, parseEnv(content), cur) {
|
||||
bench := map[string]any{
|
||||
"prefill": sb.Result.PromptPerSecond,
|
||||
"decode": sb.Result.PredictedPerSec,
|
||||
"at": sb.At,
|
||||
}
|
||||
// Types de cache à côté de la mesure : un KV quantifié n'est pas une
|
||||
// référence sans perte, la pastille le dit.
|
||||
if sb.Result.KV != "" {
|
||||
bench["kv"] = sb.Result.KV
|
||||
bench["kv_quantized"] = sb.Result.KVQuantized
|
||||
}
|
||||
if d := sb.Result.Depth; d != nil && d.DecodePerSec > 0 && d.Cold != nil {
|
||||
bench["depth"] = d.Cold.New
|
||||
bench["depth_decode"] = d.DecodePerSec
|
||||
}
|
||||
item["bench"] = bench
|
||||
}
|
||||
out = append(out, item)
|
||||
}
|
||||
@@ -1340,39 +1352,6 @@ func svcHandler(action string) http.HandlerFunc {
|
||||
// handleChat is the SSE proxy with tool-calling. The HTTP handler writes raw
|
||||
// data: lines matching what the embedded JS expects (delta.content,
|
||||
// delta.reasoning_content, delta.tool_used).
|
||||
// handleBench runs `runBench` synchronously. Long enough (~30-60s) that we
|
||||
// rely on the client side to show a spinner / disable the button.
|
||||
func handleBench(w http.ResponseWriter, r *http.Request) {
|
||||
nPrompt, nPredict := 2000, 300
|
||||
if v := r.URL.Query().Get("prompt"); v != "" {
|
||||
if parsed, err := strconv.Atoi(v); err == nil && parsed > 0 {
|
||||
nPrompt = parsed
|
||||
}
|
||||
}
|
||||
if v := r.URL.Query().Get("n"); v != "" {
|
||||
if parsed, err := strconv.Atoi(v); err == nil && parsed > 0 {
|
||||
nPredict = parsed
|
||||
}
|
||||
}
|
||||
res, err := runBench(nPrompt, nPredict)
|
||||
if err != nil {
|
||||
sendJSON(w, 500, map[string]any{"ok": false, "error": err.Error()})
|
||||
return
|
||||
}
|
||||
sendJSON(w, 200, map[string]any{"ok": true, "result": res})
|
||||
}
|
||||
|
||||
// handleBenchLast returns the most recent persisted benchmark, or {ok:false}
|
||||
// when none has been run yet.
|
||||
func handleBenchLast(w http.ResponseWriter, r *http.Request) {
|
||||
sb := loadLastBench()
|
||||
if sb == nil {
|
||||
sendJSON(w, 200, map[string]any{"ok": false})
|
||||
return
|
||||
}
|
||||
sendJSON(w, 200, map[string]any{"ok": true, "result": sb.Result, "model": sb.Model, "at": sb.At})
|
||||
}
|
||||
|
||||
// chatReq est le corps d'une requête de chat (commun au chat clair et au chat E2E).
|
||||
//
|
||||
// `messages` et `ctx_used`, que les clients envoyaient du temps où l'historique
|
||||
|
||||
@@ -330,7 +330,9 @@ func newWebMux() *http.ServeMux {
|
||||
api("/api/start", svcHandler("start"))
|
||||
api("/api/stop", svcHandler("stop"))
|
||||
api("/api/restart", svcHandler("restart"))
|
||||
api("/api/bench", handleBench)
|
||||
api("/api/bench", handleBench) // POST : lance le benchmark en arrière-plan (llm_bench_job.go)
|
||||
api("/api/bench/status", handleBenchStatus) // progression, puis résultat
|
||||
api("/api/bench/cancel", handleBenchCancel) // POST : annule
|
||||
api("/api/bench/last", handleBenchLast)
|
||||
api("/api/perf/summary", handlePerfSummary)
|
||||
api("/api/chat", handleChat) // flux d'ABONNEMENT (SSE) : rejoue + suit le fil
|
||||
|
||||
Reference in new issue
Block a user