Files
Loki/internal/loki/backend_serve_args_test.go
T
MichaelandClaude Opus 5.5 2bf388f7b7 Moteur : garde-fous de fidélité — un cache KV quantifié ou un cache approché se voit, Loki n'en pose jamais
Le cache KV en q8_0 d'un preset MoE vivait dans EXTRA_ARGS : ni l'avertissement
de lenteur ni la liste de l'interface ne le voyaient, qui affichait « f16 (max
qualité) » pendant que le moteur tournait en q8_0. Rien n'empêchait non plus un
--context-shift ou un --cache-reuse de modifier en douce ce que voit le modèle.
Aucun drapeau n'est ajouté ni retiré : on DIT ce que la ligne finale change.

- Type de cache effectif (effectiveKVTypes) : KV_TYPE*, puis -ctk/-ctv et
  --cache-type-k/v d'EXTRA_ARGS par-dessus, la dernière occurrence gagne comme
  dans llama-server. warnSlowKV le reçoit désormais.
- Note au lancement quand ce cache n'est pas f16 : q8_0 modifie légèrement les
  sorties, q4_0 perte mesurable, bf16 numérique différente. Avec -ot ou
  --n-cpu-moe, --fit ne tourne pas : repasser en f16 demandera sans doute de
  relever --n-cpu-moe. Pas d'estimation de VRAM sans les métadonnées du GGUF.
- Avertissement pour --context-shift (jamais --no-context-shift) et
  --cache-reuse N>0 ; --swa-full, sans perte, n'en déclenche aucun. Vision
  chargée : llama.cpp les ignore, on le précise.
- Interface : options du cache KV étiquetées, sous-titre qui montre le type
  effectif (« défini par EXTRA_ARGS ») et les drapeaux de cache approché.
- Tests : notes et type effectif en table, aucun -ctk/-ctv sans KV_TYPE, corps
  réels du chat et de la compaction sans cache_prompt:false ni n_cache_reuse,
  et ces clés réservées au benchmark dans tout le paquet (arbre syntaxique).

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
2026-10-03 23:04:08 +02:00

300 lines
12 KiB
Go

package loki
import (
"reflect"
"testing"
)
// Deux aides de moteur fabriquées, réduites aux lignes que lisent les tests de
// capacité : un llama.cpp récent (--load-mode, « -ngl auto », --reasoning) et
// un ancien qui ne connaît que les drapeaux d'origine.
const (
helpRecent = `-ngl, --gpu-layers, --n-gpu-layers N max. number of layers to store in VRAM, either an exact number,
'auto', or 'all' (default: auto)
-lm, --load-mode MODE how to load the model: auto, mmap, mmap+mlock, mlock, none, dio
-rea, --reasoning [on|off|auto] use reasoning/thinking in the chat (default: auto)
`
helpOld = `-ngl, --gpu-layers, --n-gpu-layers N number of layers to store in VRAM
--mlock force system to keep model in RAM
--no-mmap do not memory-map model
`
)
// TestBuildServeArgs fige la ligne de commande produite pour les presets
// courants : la moindre différence d'ordre ou de valeur casse le test. Le
// moteur retient la DERNIÈRE occurrence d'un drapeau, donc l'ordre compte
// autant que les valeurs — c'est la base sur laquelle les réglages suivants
// viendront s'ajouter, chacun avec sa ligne ici.
func TestBuildServeArgs(t *testing.T) {
const bin, model = "/opt/llama/llama-server", "/models/Qwen3.6-27B-Q4_K_M.gguf"
base := []string{bin, "-m", model,
"-c", "32768", "-b", "2048", "-ub", "512",
"--host", "0.0.0.0", "--port", "8080"}
with := func(tail ...string) []string {
return append(append([]string{}, base...), tail...)
}
// withThreads insère -t / -tb à leur place, juste après -c.
withThreads := func(threads []string, tail ...string) []string {
out := append(append([]string{}, base[:5]...), threads...)
return append(append(out, base[5:]...), tail...)
}
cases := []struct {
name string
cfg map[string]string
si serveSysInfo
want []string
wantEnv map[string]string
wantNotes int
}{
{
name: "dense par défaut, moteur récent : 999 devient auto, et on le dit",
cfg: map[string]string{"NGL": "999"},
si: serveSysInfo{Help: helpRecent},
want: with("--parallel", "1", "-ngl", "auto"), wantNotes: 1,
},
{
name: "dense par défaut, moteur ancien : 999 reste 999",
cfg: map[string]string{"NGL": "999"},
si: serveSysInfo{Help: helpOld},
want: with("--parallel", "1", "-ngl", "999"),
},
{
name: "clé NGL absente, aide illisible : forme historique",
cfg: map[string]string{},
si: serveSysInfo{},
want: with("--parallel", "1", "-ngl", "999"),
},
{
name: "NGL forcé à un nombre : jamais touché",
cfg: map[string]string{"NGL": "28"},
si: serveSysInfo{Help: helpRecent},
want: with("--parallel", "1", "-ngl", "28"),
},
{
name: "NGL=auto : aucun drapeau",
cfg: map[string]string{"NGL": "auto"},
si: serveSysInfo{Help: helpRecent},
want: with("--parallel", "1"),
},
{
name: "réglages du preset et cache KV commun",
cfg: map[string]string{"CTX": "65536", "THREADS": "8", "THREADS_BATCH": "16", "BATCH": "4096",
"UBATCH": "1024", "HOST": "127.0.0.1", "PORT": "9090", "PARALLEL": "2", "NGL": "all",
"KV_TYPE": "q8_0"},
si: serveSysInfo{Help: helpRecent},
want: []string{bin, "-m", model,
"-c", "65536", "-t", "8", "-tb", "16", "-b", "4096", "-ub", "1024",
"--host", "127.0.0.1", "--port", "9090",
"--parallel", "2", "-ngl", "all", "-ctk", "q8_0", "-ctv", "q8_0"},
wantNotes: 1, // cache quantifié : dit, jamais retiré
},
{
name: "THREADS=0 et THREADS_BATCH=0 : aucun drapeau, llama.cpp prend ses cœurs physiques",
cfg: map[string]string{"NGL": "28", "THREADS": "0", "THREADS_BATCH": "0"},
si: serveSysInfo{Help: helpRecent},
want: with("--parallel", "1", "-ngl", "28"),
},
{
name: "THREADS=6 seul : -t 6, -tb laissé au moteur (il recopie -t)",
cfg: map[string]string{"NGL": "28", "THREADS": "6"},
si: serveSysInfo{Help: helpRecent},
want: withThreads([]string{"-t", "6"}, "--parallel", "1", "-ngl", "28"),
},
{
name: "THREADS_BATCH seul : -tb sans -t",
cfg: map[string]string{"NGL": "28", "THREADS_BATCH": "12"},
si: serveSysInfo{Help: helpRecent},
want: withThreads([]string{"-tb", "12"}, "--parallel", "1", "-ngl", "28"),
},
{
name: "THREADS illisible ou négatif : ignoré, et dit",
cfg: map[string]string{"NGL": "28", "THREADS": "auto", "THREADS_BATCH": "-1"},
si: serveSysInfo{Help: helpRecent},
want: with("--parallel", "1", "-ngl", "28"), wantNotes: 2,
},
{
name: "-t=6 dans EXTRA_ARGS : pas de -t de Loki",
cfg: map[string]string{"NGL": "28", "THREADS": "8", "THREADS_BATCH": "16", "EXTRA_ARGS": "-t=6 --threads-batch 10"},
si: serveSysInfo{Help: helpRecent},
want: with("--parallel", "1", "-ngl", "28", "-t=6", "--threads-batch", "10"),
},
{
// Sans -tb, llama.cpp remplacerait tout le réglage CPU du batch par
// celui de -t : le --poll-batch de l'utilisateur serait effacé.
name: "--poll-batch dans EXTRA_ARGS, THREADS vide : -tb 0 le protège, et on le dit",
cfg: map[string]string{"NGL": "28", "EXTRA_ARGS": "--poll-batch 0"},
si: serveSysInfo{Help: helpRecent},
want: withThreads([]string{"-tb", "0"}, "--parallel", "1", "-ngl", "28", "--poll-batch", "0"), wantNotes: 1,
},
{
name: "-Cb dans EXTRA_ARGS, THREADS=6 : -tb recopie 6",
cfg: map[string]string{"NGL": "28", "THREADS": "6", "EXTRA_ARGS": "-Cb 0xff"},
si: serveSysInfo{Help: helpRecent},
want: withThreads([]string{"-t", "6", "-tb", "6"}, "--parallel", "1", "-ngl", "28", "-Cb", "0xff"),
},
{
name: "-t=5 et --prio-batch dans EXTRA_ARGS : -tb recopie 5",
cfg: map[string]string{"NGL": "28", "EXTRA_ARGS": "-t=5 --prio-batch 2"},
si: serveSysInfo{Help: helpRecent},
want: withThreads([]string{"-tb", "5"}, "--parallel", "1", "-ngl", "28", "-t=5", "--prio-batch", "2"),
},
{
name: "-Crb et -tb dans EXTRA_ARGS : Loki se tait",
cfg: map[string]string{"NGL": "28", "EXTRA_ARGS": "-Crb 0-7 -tb 8"},
si: serveSysInfo{Help: helpRecent},
want: with("--parallel", "1", "-ngl", "28", "-Crb", "0-7", "-tb", "8"),
},
{
name: "conteneur à l'étroit : -t de la sonde, et une note",
cfg: map[string]string{"NGL": "28"},
si: serveSysInfo{Help: helpRecent, CPU: cpuBudget{N: 4, Engine: 8, Why: "quota cgroup de 4 CPU"}},
want: withThreads([]string{"-t", "4"}, "--parallel", "1", "-ngl", "28"), wantNotes: 1,
},
{
name: "THREADS explicite : la sonde se tait",
cfg: map[string]string{"NGL": "28", "THREADS": "6"},
si: serveSysInfo{Help: helpRecent, CPU: cpuBudget{N: 4, Engine: 8, Why: "quota"}},
want: withThreads([]string{"-t", "6"}, "--parallel", "1", "-ngl", "28"),
},
{
name: "masque d'affinité dans EXTRA_ARGS : la sonde se tait",
cfg: map[string]string{"NGL": "28", "EXTRA_ARGS": "-Cr 0-3"},
si: serveSysInfo{Help: helpRecent, CPU: cpuBudget{N: 4, Engine: 8, Why: "quota"}},
want: with("--parallel", "1", "-ngl", "28", "-Cr", "0-3"),
},
{
name: "cache KV séparé K/V",
cfg: map[string]string{"NGL": "28", "KV_TYPE": "q8_0", "KV_TYPE_V": "q4_0"},
si: serveSysInfo{Help: helpRecent},
want: with("--parallel", "1", "-ngl", "28", "-ctk", "q8_0", "-ctv", "q4_0"), wantNotes: 1,
},
{
// Preset MoE typique : experts sur CPU via EXTRA_ARGS, qui pose aussi
// --parallel, -ngl et le cache KV. Loki n'ajoute alors ni --parallel ni
// -ngl, et EXTRA_ARGS ferme la marche (la dernière occurrence gagne).
name: "MoE : EXTRA_ARGS avec -ot, --n-cpu-moe, -ctk q8_0",
cfg: map[string]string{"NGL": "999", "PARALLEL": "1",
"EXTRA_ARGS": `-ngl 99 --parallel 1 -ot "blk\.(\d+)\.ffn_.*_exps=CPU" --n-cpu-moe 30 -ctk q8_0 -ctv q8_0 -fa on`},
si: serveSysInfo{Help: helpRecent},
want: with("-ngl", "99", "--parallel", "1", "-ot", `blk\.(\d+)\.ffn_.*_exps=CPU`,
"--n-cpu-moe", "30", "-ctk", "q8_0", "-ctv", "q8_0", "-fa", "on"),
wantNotes: 1, // q8_0 venu d'EXTRA_ARGS : vu comme celui de KV_TYPE
},
{
name: "--n-gpu-layers=N dans EXTRA_ARGS : pas de -ngl en double",
cfg: map[string]string{"NGL": "999", "EXTRA_ARGS": "--n-gpu-layers=40 -np 2"},
si: serveSysInfo{Help: helpRecent},
want: with("--n-gpu-layers=40", "-np", "2"),
},
{
name: "raisonnement actif : budget illimité par défaut",
cfg: map[string]string{"NGL": "28", "REASONING": "on"},
si: serveSysInfo{Help: helpRecent},
want: with("--parallel", "1", "-ngl", "28", "--reasoning", "on", "--reasoning-budget", "-1"),
},
{
name: "raisonnement auto avec budget choisi",
cfg: map[string]string{"NGL": "28", "REASONING": " auto ", "REASONING_BUDGET": "4096"},
si: serveSysInfo{Help: helpRecent},
want: with("--parallel", "1", "-ngl", "28", "--reasoning", "auto", "--reasoning-budget", "4096"),
},
{
name: "raisonnement interdit, moteur qui connaît le drapeau",
cfg: map[string]string{"NGL": "28", "REASONING": "off"},
si: serveSysInfo{Help: helpRecent},
want: with("--parallel", "1", "-ngl", "28", "--reasoning", "off"),
},
{
name: "raisonnement interdit, vieux moteur : aucun drapeau, une note",
cfg: map[string]string{"NGL": "28", "REASONING": "off"},
si: serveSysInfo{Help: helpOld},
want: with("--parallel", "1", "-ngl", "28"), wantNotes: 1,
},
{
name: "vision et clé d'API, avant EXTRA_ARGS",
cfg: map[string]string{"NGL": "28", "EXTRA_ARGS": "-fa on"},
si: serveSysInfo{Help: helpRecent, MMProj: "/models/mmproj-F16.gguf", APIKey: "secret"},
want: with("--parallel", "1", "-ngl", "28", "--mmproj", "/models/mmproj-F16.gguf",
"--api-key", "secret", "-fa", "on"),
},
{
name: "drapeaux de chargement anciens → --load-mode sur moteur récent",
cfg: map[string]string{"NGL": "28", "EXTRA_ARGS": "--mlock --no-mmap -fa on"},
si: serveSysInfo{Help: helpRecent},
want: with("--parallel", "1", "-ngl", "28", "-fa", "on", "--load-mode", "mlock"),
},
{
name: "--load-mode retraduit pour un moteur ancien",
cfg: map[string]string{"NGL": "28", "EXTRA_ARGS": "--load-mode none"},
si: serveSysInfo{Help: helpOld},
want: with("--parallel", "1", "-ngl", "28", "--no-mmap"),
},
{
// dio n'existe pas sur l'ancien moteur : retiré, et dit par une note
// RENDUE (buildServeArgs n'écrit rien elle-même).
name: "--load-mode dio sur moteur ancien : retiré, une note",
cfg: map[string]string{"NGL": "28", "EXTRA_ARGS": "--load-mode dio -fa on"},
si: serveSysInfo{Help: helpOld},
want: with("--parallel", "1", "-ngl", "28", "-fa", "on"), wantNotes: 1,
},
{
name: "sélection GPU : variables d'environnement, ligne inchangée",
cfg: map[string]string{"NGL": "28", "CUDA_VISIBLE_DEVICES": "1,0"},
si: serveSysInfo{Help: helpRecent},
want: with("--parallel", "1", "-ngl", "28"),
wantEnv: map[string]string{"CUDA_VISIBLE_DEVICES": "1,0", "CUDA_DEVICE_ORDER": "PCI_BUS_ID"},
},
}
for _, c := range cases {
t.Run(c.name, func(t *testing.T) {
c.si.Model = model
args, env, notes := buildServeArgs(c.cfg, splitArgs(c.cfg["EXTRA_ARGS"]), bin, c.si)
if !reflect.DeepEqual(args, c.want) {
t.Fatalf("args\n got %q\nwant %q", args, c.want)
}
wantEnv := c.wantEnv
if wantEnv == nil {
wantEnv = map[string]string{}
}
if !reflect.DeepEqual(env, wantEnv) {
t.Fatalf("env = %v, want %v", env, wantEnv)
}
if len(notes) != c.wantNotes {
t.Fatalf("notes = %q, en attendait %d", notes, c.wantNotes)
}
})
}
}
func TestHelpCapabilities(t *testing.T) {
if !helpSupportsReasoningFlag(helpRecent) || helpSupportsReasoningFlag(helpOld) {
t.Fatal("--reasoning mal détecté")
}
if !helpSupportsLoadMode(helpRecent) || helpSupportsLoadMode(helpOld) {
t.Fatal("--load-mode mal détecté")
}
if !helpSupportsNGLAuto(helpRecent) || helpSupportsNGLAuto(helpOld) {
t.Fatal("« -ngl auto » mal détecté")
}
// Aide illisible = aucune capacité : on garde la forme historique.
if helpFitsLayersItself("") || helpSupportsReasoningFlag("") {
t.Fatal("une aide vide ne doit promettre aucune capacité")
}
}
func TestServeKVTypes(t *testing.T) {
for _, c := range []struct {
cfg map[string]string
k, v string
}{
{map[string]string{}, "", ""},
{map[string]string{"KV_TYPE": "q8_0"}, "q8_0", "q8_0"},
{map[string]string{"KV_TYPE": "q8_0", "KV_TYPE_K": "f16"}, "f16", "q8_0"},
{map[string]string{"KV_TYPE_V": "q4_0"}, "", "q4_0"},
} {
if k, v := serveKVTypes(c.cfg); k != c.k || v != c.v {
t.Errorf("%v : got %s/%s, want %s/%s", c.cfg, k, v, c.k, c.v)
}
}
}