diff --git a/internal/loki/backend_serve.go b/internal/loki/backend_serve.go index 364480c..9d86e2d 100644 --- a/internal/loki/backend_serve.go +++ b/internal/loki/backend_serve.go @@ -8,6 +8,7 @@ import ( "os" "os/exec" "path/filepath" + "runtime" "strings" "sync" "time" @@ -328,6 +329,12 @@ func cmdServe(args []string) error { // $LOKI_HOME/.api_key en priorité (elle survit ainsi aux changements de preset // qui réécrivent config.env), avec config.env comme repli rétro-compatible. si.APIKey, _ = effectiveAPIKeyErr() + // Conteneur à l'étroit (cpuset, quota) : seul Linux l'expose, dans /proc et + // /sys. Un LLAMA_ARG_THREADS déjà posé est un choix de l'utilisateur — un -t + // de Loki l'écraserait, donc on ne sonde même pas. + if runtime.GOOS == "linux" && os.Getenv("LLAMA_ARG_THREADS") == "" { + si.CPU = cpusetThreads(os.DirFS("/")) + } llmArgs, env, notes := buildServeArgs(cfg, splitArgs(cfg["EXTRA_ARGS"]), bin, si) applyServeEnv(env) @@ -361,10 +368,11 @@ func cmdServe(args []string) error { // capacités), les chemins déjà résolus et vérifiés, la clé d'API. Avec elle, // buildServeArgs n'a plus besoin de toucher ni au disque ni au binaire. type serveSysInfo struct { - Help string // sortie de « --help » ; vide = moteur inconnu, aucun risque pris - Model string // chemin du .gguf principal, résolu et vérifié - MMProj string // chemin du projecteur vision, résolu et vérifié ; vide = pas de vision - APIKey string // clé effective (.api_key, sinon config.env) ; vide = serveur ouvert + Help string // sortie de « --help » ; vide = moteur inconnu, aucun risque pris + Model string // chemin du .gguf principal, résolu et vérifié + MMProj string // chemin du projecteur vision, résolu et vérifié ; vide = pas de vision + APIKey string // clé effective (.api_key, sinon config.env) ; vide = serveur ouvert + CPU cpuBudget // sonde de conteneur (Linux) ; zéro = llama.cpp choisit ses threads seul } // cudaDeviceEnv : sélection GPU (loki gpu), on filtre les devices visibles par @@ -426,16 +434,21 @@ func buildServeArgs(cfg map[string]string, extra []string, bin string, si serveS notes = append(notes, loadNote) } + // Threads : vide ou 0 = AUCUN drapeau, pour que llama.cpp prenne ses cœurs + // physiques au lieu de tous les threads logiques (voir threadArgs). + threads, threadNotes := threadArgs(cfg["THREADS"], cfg["THREADS_BATCH"], extra, si.CPU) + notes = append(notes, threadNotes...) args = []string{bin, "-m", si.Model, "-c", get("CTX", "32768"), - "-t", get("THREADS", "0"), - "-tb", get("THREADS_BATCH", "0"), + } + args = append(args, threads...) + args = append(args, "-b", get("BATCH", "2048"), "-ub", get("UBATCH", "512"), "--host", get("HOST", "0.0.0.0"), "--port", get("PORT", "8080"), - } + ) // --parallel 1 EXPLICITE : les llama-server récents ouvrent plusieurs slots // par défaut, soit des tampons de calcul GPU multipliés d'autant — pour un // serveur mono-utilisateur comme Loki, c'est de la VRAM brûlée pour rien. diff --git a/internal/loki/backend_serve_args_test.go b/internal/loki/backend_serve_args_test.go index a98dc44..e3a3cb4 100644 --- a/internal/loki/backend_serve_args_test.go +++ b/internal/loki/backend_serve_args_test.go @@ -28,11 +28,16 @@ const ( func TestBuildServeArgs(t *testing.T) { const bin, model = "/opt/llama/llama-server", "/models/Qwen3.6-27B-Q4_K_M.gguf" base := []string{bin, "-m", model, - "-c", "32768", "-t", "0", "-tb", "0", "-b", "2048", "-ub", "512", + "-c", "32768", "-b", "2048", "-ub", "512", "--host", "0.0.0.0", "--port", "8080"} with := func(tail ...string) []string { return append(append([]string{}, base...), tail...) } + // withThreads insère -t / -tb à leur place, juste après -c. + withThreads := func(threads []string, tail ...string) []string { + out := append(append([]string{}, base[:5]...), threads...) + return append(append(out, base[5:]...), tail...) + } cases := []struct { name string cfg map[string]string @@ -82,6 +87,54 @@ func TestBuildServeArgs(t *testing.T) { "--host", "127.0.0.1", "--port", "9090", "--parallel", "2", "-ngl", "all", "-ctk", "q8_0", "-ctv", "q8_0"}, }, + { + name: "THREADS=0 et THREADS_BATCH=0 : aucun drapeau, llama.cpp prend ses cœurs physiques", + cfg: map[string]string{"NGL": "28", "THREADS": "0", "THREADS_BATCH": "0"}, + si: serveSysInfo{Help: helpRecent}, + want: with("--parallel", "1", "-ngl", "28"), + }, + { + name: "THREADS=6 seul : -t 6, -tb laissé au moteur (il recopie -t)", + cfg: map[string]string{"NGL": "28", "THREADS": "6"}, + si: serveSysInfo{Help: helpRecent}, + want: withThreads([]string{"-t", "6"}, "--parallel", "1", "-ngl", "28"), + }, + { + name: "THREADS_BATCH seul : -tb sans -t", + cfg: map[string]string{"NGL": "28", "THREADS_BATCH": "12"}, + si: serveSysInfo{Help: helpRecent}, + want: withThreads([]string{"-tb", "12"}, "--parallel", "1", "-ngl", "28"), + }, + { + name: "THREADS illisible ou négatif : ignoré, et dit", + cfg: map[string]string{"NGL": "28", "THREADS": "auto", "THREADS_BATCH": "-1"}, + si: serveSysInfo{Help: helpRecent}, + want: with("--parallel", "1", "-ngl", "28"), wantNotes: 2, + }, + { + name: "-t=6 dans EXTRA_ARGS : pas de -t de Loki", + cfg: map[string]string{"NGL": "28", "THREADS": "8", "THREADS_BATCH": "16", "EXTRA_ARGS": "-t=6 --threads-batch 10"}, + si: serveSysInfo{Help: helpRecent}, + want: with("--parallel", "1", "-ngl", "28", "-t=6", "--threads-batch", "10"), + }, + { + name: "conteneur à l'étroit : -t de la sonde, et une note", + cfg: map[string]string{"NGL": "28"}, + si: serveSysInfo{Help: helpRecent, CPU: cpuBudget{N: 4, Engine: 8, Why: "quota cgroup de 4 CPU"}}, + want: withThreads([]string{"-t", "4"}, "--parallel", "1", "-ngl", "28"), wantNotes: 1, + }, + { + name: "THREADS explicite : la sonde se tait", + cfg: map[string]string{"NGL": "28", "THREADS": "6"}, + si: serveSysInfo{Help: helpRecent, CPU: cpuBudget{N: 4, Engine: 8, Why: "quota"}}, + want: withThreads([]string{"-t", "6"}, "--parallel", "1", "-ngl", "28"), + }, + { + name: "masque d'affinité dans EXTRA_ARGS : la sonde se tait", + cfg: map[string]string{"NGL": "28", "EXTRA_ARGS": "-Cr 0-3"}, + si: serveSysInfo{Help: helpRecent, CPU: cpuBudget{N: 4, Engine: 8, Why: "quota"}}, + want: with("--parallel", "1", "-ngl", "28", "-Cr", "0-3"), + }, { name: "cache KV séparé K/V", cfg: map[string]string{"NGL": "28", "KV_TYPE": "q8_0", "KV_TYPE_V": "q4_0"}, diff --git a/internal/loki/backend_serve_threads.go b/internal/loki/backend_serve_threads.go new file mode 100644 index 0000000..3b57880 --- /dev/null +++ b/internal/loki/backend_serve_threads.go @@ -0,0 +1,306 @@ +package loki + +import ( + "fmt" + "io/fs" + "sort" + "strconv" + "strings" +) + +// Threads CPU de llama-server. +// +// Loki passait toujours « -t -tb ». Or pour +// llama.cpp, 0 (ou moins) veut dire std::thread::hardware_concurrency() : TOUS +// les threads logiques, frères SMT compris. Avec l'attente active par défaut +// (--poll 50), deux threads qui se disputent le même cœur physique se gênent +// plus qu'ils ne s'aident — c'est le décodage des experts MoE sur CPU +// (--n-cpu-moe) qui paie. Le « 0 = auto » de l'interface mentait donc : ce +// n'était pas l'auto de llama.cpp. +// +// Le vrai auto, c'est l'ABSENCE du drapeau : n_threads reste à -1 et le moteur +// prend common_cpu_get_num_math() — les cœurs physiques (sans les cœurs E sur +// Intel hybride, sous Linux x86). Un -tb absent recopie -t, et le brouillon +// spéculatif hérite des deux. Le nombre de threads ne change que la répartition +// du travail, pas le calcul du modèle. +// +// Effet de bord assumé : sans -t de Loki, une variable LLAMA_ARG_THREADS posée +// dans l'environnement du moteur s'applique enfin (le -t 0 l'écrasait). +// +// Gain non mesuré sur la machine de référence : avant d'en revendiquer un, +// comparer tg du preset MoE à physiques, physiques-1 et logiques (MTP actif et +// coupé, la vérification spéculative passant par n_threads_batch). Si les +// logiques gagnent, poser THREADS dans CE preset plutôt que changer le défaut. +// Pour voir le nombre retenu : -lv 4 dans EXTRA_ARGS (system_info est en TRACE). + +// cpuBudget est ce que la sonde de conteneur (cpusetThreads) a conclu. N = 0 : +// rien à dire, llama.cpp choisit seul. +type cpuBudget struct { + N int // threads à imposer via -t + Engine int // ce que llama.cpp aurait pris seul, pour la note + Why string // raison, en clair, pour stderr +} + +// threadCount lit THREADS / THREADS_BATCH. Vide ou 0 : auto, sans un mot. Une +// valeur illisible (« auto ») ou négative faisait jusqu'ici tomber le moteur en +// boucle sur son analyse d'arguments ; elle est désormais ignorée — mais on le +// DIT, plutôt que de la jeter en douce. +func threadCount(key, v string) (int, string) { + v = strings.TrimSpace(v) + if v == "" { + return 0, "" + } + n, err := strconv.Atoi(v) + if err != nil || n < 0 { + return 0, fmt.Sprintf("%s=%s ignoré (pas un nombre de threads ≥ 0) → automatique", key, v) + } + return n, "" +} + +// threadArgs compose -t / -tb. Fonction pure, comme nglArgs : la sonde du +// conteneur arrive déjà tranchée dans auto. +// +// THREADS=N>0 → -t N +// THREADS vide/0, sonde N>0 → -t N (conteneur à l'étroit, voir cpusetThreads) +// THREADS vide/0 → rien (cœurs physiques, choisis par llama.cpp) +// THREADS_BATCH=N>0 → -tb N +// THREADS_BATCH vide/0 → rien (le moteur recopie -t) +// +// Un -t / -tb déjà écrit dans EXTRA_ARGS gagne : on ne double pas le drapeau. +// Un masque d'affinité posé à la main (-C, -Cr, --cpu-strict) désarme la sonde : +// l'utilisateur a pris la main sur le placement, on ne devine rien par-dessus. +func threadArgs(threads, threadsBatch string, extra []string, auto cpuBudget) (args, notes []string) { + t, note := threadCount("THREADS", threads) + if note != "" { + notes = append(notes, note) + } + switch { + case hasAnyFlag(extra, "-t", "--threads"): + case t > 0: + args = append(args, "-t", strconv.Itoa(t)) + case auto.N > 0 && !hasAnyFlag(extra, "-C", "--cpu-mask", "-Cr", "--cpu-range", "--cpu-strict"): + args = append(args, "-t", strconv.Itoa(auto.N)) + notes = append(notes, fmt.Sprintf("threads auto → -t %d : %s (llama.cpp seul en lancerait %d)", + auto.N, auto.Why, auto.Engine)) + } + tb, note := threadCount("THREADS_BATCH", threadsBatch) + if note != "" { + notes = append(notes, note) + } + if tb > 0 && !hasAnyFlag(extra, "-tb", "--threads-batch") { + args = append(args, "-tb", strconv.Itoa(tb)) + } + return args, notes +} + +// cpusetThreads : sonde de conteneur, Linux seulement (fsys = la racine « / »). +// +// L'auto de llama.cpp se trompe quand le conteneur n'a droit qu'à une partie +// de la machine : +// - cpuset restreint (docker --cpuset-cpus) : common_cpu_get_num_math essaie +// d'épingler un thread sur chaque CPU de l'hôte, échoue hors du cpuset et +// se rabat sur TOUS les cœurs physiques de l'hôte (sysfs n'est pas cloisonné), +// cœurs E compris ; +// - quota CFS (docker --cpus) : invisible dans l'affinité, il étrangle des +// threads qui attendent en boucle active — le pire cas. +// +// On compte alors comme llama.cpp (chaînes thread_siblings distinctes, cœurs E +// de /sys/devices/cpu_atom/cpus écartés), mais sur les seuls CPU permis, plafonné +// par le quota. On ne renvoie un nombre que s'il est PLUS PETIT que ce que le +// moteur aurait pris seul ; à la moindre lecture ratée, rien (le moteur décide). +func cpusetThreads(fsys fs.FS) cpuBudget { + online, err := readCPUList(fsys, "sys/devices/system/cpu/online") + if err != nil || len(online) == 0 { + return cpuBudget{} + } + allowed, err := procAllowedCPUs(fsys) + if err != nil { + return cpuBudget{} + } + allowed = intersectCPUs(allowed, online) + if len(allowed) == 0 { + return cpuBudget{} + } + restricted := len(allowed) < len(online) + quota := cgroupCPUQuota(fsys) + if !restricted && quota == 0 { + return cpuBudget{} + } + // Absent = CPU non hybride : aucun cœur E à écarter. + atom, _ := readCPUList(fsys, "sys/devices/cpu_atom/cpus") + + hostAll, err := physicalCores(fsys, online, nil) + if err != nil { + return cpuBudget{} + } + hostMath, err := physicalCores(fsys, online, atom) + if err != nil { + return cpuBudget{} + } + if hostMath == 0 { + hostMath = hostAll + } + + var n, engine int + var why string + if restricted { + engine = hostAll // épinglage refusé hors cpuset → repli sur tous les cœurs de l'hôte + if n, err = physicalCores(fsys, allowed, atom); err != nil { + return cpuBudget{} + } + if n == 0 { // conteneur cantonné aux cœurs E : on les prend, faute de mieux + if n, err = physicalCores(fsys, allowed, nil); err != nil { + return cpuBudget{} + } + } + why = fmt.Sprintf("cpuset du conteneur, %d cœurs physiques permis sur %d", n, hostAll) + if len(atom) > 0 { + why += " (cœurs E écartés)" + } + } else { + engine, n = hostMath, hostMath + } + if quota > 0 && quota < n { + n = quota + why = fmt.Sprintf("quota cgroup de %d CPU", quota) + } + if n <= 0 || n >= engine { + return cpuBudget{} + } + return cpuBudget{N: n, Engine: engine, Why: why} +} + +// physicalCores compte les cœurs physiques parmi cpus comme le fait +// common_cpu_get_num_physical_cores : une chaîne thread_siblings distincte par +// cœur — juste aussi sur les machines multi-puces, où (package, core_id) peut +// se répéter. Les CPU de exclude (cœurs E) ne comptent pas. +func physicalCores(fsys fs.FS, cpus, exclude []int) (int, error) { + skip := map[int]bool{} + for _, c := range exclude { + skip[c] = true + } + seen := map[string]bool{} + for _, c := range cpus { + if skip[c] { + continue + } + b, err := fs.ReadFile(fsys, fmt.Sprintf("sys/devices/system/cpu/cpu%d/topology/thread_siblings", c)) + if err != nil { + return 0, err + } + seen[strings.TrimSpace(string(b))] = true + } + return len(seen), nil +} + +// procAllowedCPUs lit Cpus_allowed_list dans /proc/self/status : l'affinité +// réelle du processus, cpuset du conteneur compris. +func procAllowedCPUs(fsys fs.FS) ([]int, error) { + b, err := fs.ReadFile(fsys, "proc/self/status") + if err != nil { + return nil, err + } + for _, line := range strings.Split(string(b), "\n") { + if v, ok := strings.CutPrefix(line, "Cpus_allowed_list:"); ok { + return parseCPUList(v) + } + } + return nil, fmt.Errorf("pas de Cpus_allowed_list") +} + +// cgroupCPUQuota renvoie le quota CPU du cgroup arrondi au CPU supérieur +// (au moins 1), 0 s'il n'y en a pas ou s'il est illisible. cgroup v2 +// (cpu.max : « max 100000 » ou « 250000 100000 »), sinon v1 (cfs_quota_us, +// -1 = illimité). Seul un conteneur voit son propre cgroup sous /sys/fs/cgroup. +func cgroupCPUQuota(fsys fs.FS) int { + if b, err := fs.ReadFile(fsys, "sys/fs/cgroup/cpu.max"); err == nil { + f := strings.Fields(string(b)) + if len(f) == 0 || f[0] == "max" { + return 0 + } + period := "100000" + if len(f) > 1 { + period = f[1] + } + return ceilQuota(f[0], period) + } + for _, dir := range []string{"sys/fs/cgroup/cpu", "sys/fs/cgroup/cpu,cpuacct"} { + q, err := fs.ReadFile(fsys, dir+"/cpu.cfs_quota_us") + if err != nil { + continue + } + p, err := fs.ReadFile(fsys, dir+"/cpu.cfs_period_us") + if err != nil { + return 0 + } + return ceilQuota(strings.TrimSpace(string(q)), strings.TrimSpace(string(p))) + } + return 0 +} + +func ceilQuota(quota, period string) int { + q, err1 := strconv.ParseInt(quota, 10, 64) + p, err2 := strconv.ParseInt(period, 10, 64) + if err1 != nil || err2 != nil || q <= 0 || p <= 0 { + return 0 + } + return int(max(1, (q+p-1)/p)) +} + +// readCPUList lit un fichier au format liste du noyau (« 0-3,8-11 »). +func readCPUList(fsys fs.FS, name string) ([]int, error) { + b, err := fs.ReadFile(fsys, name) + if err != nil { + return nil, err + } + return parseCPUList(string(b)) +} + +// parseCPUList décode « 0-3,8-11 » en liste triée, sans doublon. Une liste +// vide est valide (aucun CPU) ; une syntaxe inconnue est une erreur. +func parseCPUList(s string) ([]int, error) { + set := map[int]bool{} + for _, part := range strings.Split(strings.TrimSpace(s), ",") { + part = strings.TrimSpace(part) + if part == "" { + continue + } + lo, hi, isRange := strings.Cut(part, "-") + a, err := strconv.Atoi(lo) + if err != nil { + return nil, err + } + b := a + if isRange { + if b, err = strconv.Atoi(hi); err != nil { + return nil, err + } + } + if a < 0 || b < a || b-a > 1<<16 { + return nil, fmt.Errorf("liste de CPU invalide : %q", part) + } + for c := a; c <= b; c++ { + set[c] = true + } + } + out := make([]int, 0, len(set)) + for c := range set { + out = append(out, c) + } + sort.Ints(out) + return out, nil +} + +func intersectCPUs(a, b []int) []int { + in := map[int]bool{} + for _, c := range b { + in[c] = true + } + var out []int + for _, c := range a { + if in[c] { + out = append(out, c) + } + } + return out +} diff --git a/internal/loki/backend_serve_threads_test.go b/internal/loki/backend_serve_threads_test.go new file mode 100644 index 0000000..5c9c2c0 --- /dev/null +++ b/internal/loki/backend_serve_threads_test.go @@ -0,0 +1,145 @@ +package loki + +import ( + "fmt" + "reflect" + "testing" + "testing/fstest" +) + +// cpuFS fabrique un /proc + /sys minimal : online, l'affinité du processus et, +// pour chaque CPU en ligne, sa chaîne thread_siblings (siblings[cpu]). +func cpuFS(online, allowed string, siblings map[int]string, extra map[string]string) fstest.MapFS { + m := fstest.MapFS{ + "sys/devices/system/cpu/online": {Data: []byte(online + "\n")}, + "proc/self/status": {Data: []byte("Name:\tllama-server\nCpus_allowed:\tffff\nCpus_allowed_list:\t" + allowed + "\nMems_allowed:\t1\n")}, + } + for c, s := range siblings { + m[fmt.Sprintf("sys/devices/system/cpu/cpu%d/topology/thread_siblings", c)] = &fstest.MapFile{Data: []byte(s + "\n")} + } + for k, v := range extra { + m[k] = &fstest.MapFile{Data: []byte(v)} + } + return m +} + +// smtPairs : n cœurs à deux threads, frères adjacents (0-1, 2-3…) — la +// numérotation que suppose cpu_count_math_cpus. +func smtPairs(cores, from int) map[int]string { + m := map[int]string{} + for i := 0; i < cores; i++ { + s := fmt.Sprintf("core%d", from/2+i) + m[from+2*i], m[from+2*i+1] = s, s + } + return m +} + +// hybrid : 8 cœurs P à deux threads (CPU 0-15) + 8 cœurs E (CPU 16-23). +func hybrid() map[int]string { + m := smtPairs(8, 0) + for c := 16; c < 24; c++ { + m[c] = fmt.Sprintf("ecore%d", c) + } + return m +} + +func TestCPUSetThreads(t *testing.T) { + cases := []struct { + name string + fs fstest.MapFS + n int + }{ + {"machine entière, sans quota : rien", cpuFS("0-15", "0-15", smtPairs(8, 0), nil), 0}, + {"cpuset de 8 threads SMT : 4 cœurs", cpuFS("0-15", "0-7", smtPairs(8, 0), nil), 4}, + { + // L'exemple qui tournait mal : tout sauf cpu0 sur un Intel hybride. + // Les cœurs E sont écartés, cpu1 partage son cœur avec cpu0 → 8, pas 15. + "hybride, tout sauf cpu0 : 8 cœurs P", + cpuFS("0-23", "1-23", hybrid(), map[string]string{"sys/devices/cpu_atom/cpus": "16-23\n"}), 8, + }, + { + "hybride, cantonné aux cœurs E : on les prend", + cpuFS("0-23", "16-19", hybrid(), map[string]string{"sys/devices/cpu_atom/cpus": "16-23\n"}), 4, + }, + { + "quota v2 de 2,5 CPU : arrondi à 3", + cpuFS("0-15", "0-15", smtPairs(8, 0), map[string]string{"sys/fs/cgroup/cpu.max": "250000 100000\n"}), 3, + }, + { + "quota v2 illimité : rien", + cpuFS("0-15", "0-15", smtPairs(8, 0), map[string]string{"sys/fs/cgroup/cpu.max": "max 100000\n"}), 0, + }, + { + "quota plus large que les cœurs : rien", + cpuFS("0-15", "0-15", smtPairs(8, 0), map[string]string{"sys/fs/cgroup/cpu.max": "1200000 100000\n"}), 0, + }, + { + "quota v1 de 2 CPU", + cpuFS("0-15", "0-15", smtPairs(8, 0), map[string]string{ + "sys/fs/cgroup/cpu,cpuacct/cpu.cfs_quota_us": "200000\n", + "sys/fs/cgroup/cpu,cpuacct/cpu.cfs_period_us": "100000\n"}), 2, + }, + { + "quota v1 illimité (-1) : rien", + cpuFS("0-15", "0-15", smtPairs(8, 0), map[string]string{ + "sys/fs/cgroup/cpu/cpu.cfs_quota_us": "-1\n", + "sys/fs/cgroup/cpu/cpu.cfs_period_us": "100000\n"}), 0, + }, + { + "cpuset ET quota : le plus petit", + cpuFS("0-15", "0-7", smtPairs(8, 0), map[string]string{"sys/fs/cgroup/cpu.max": "200000 100000"}), 2, + }, + { + "topologie illisible : on ne devine pas", + cpuFS("0-15", "0-7", map[int]string{0: "core0"}, nil), 0, + }, + {"pas de /proc : rien", fstest.MapFS{"sys/devices/system/cpu/online": {Data: []byte("0-3")}}, 0}, + {"liste illisible : rien", cpuFS("0-15", "0-x", smtPairs(8, 0), nil), 0}, + } + for _, c := range cases { + t.Run(c.name, func(t *testing.T) { + got := cpusetThreads(c.fs) + if got.N != c.n { + t.Fatalf("N = %d (%+v), want %d", got.N, got, c.n) + } + if got.N > 0 && (got.Why == "" || got.Engine <= got.N) { + t.Fatalf("budget incomplet : %+v", got) + } + }) + } +} + +func TestParseCPUList(t *testing.T) { + for _, c := range []struct { + in string + want []int + ok bool + }{ + {"0-3,8-9\n", []int{0, 1, 2, 3, 8, 9}, true}, + {" 5 ", []int{5}, true}, + {"2,0-2", []int{0, 1, 2}, true}, + {"", []int{}, true}, + {"3-1", nil, false}, + {"a", nil, false}, + } { + got, err := parseCPUList(c.in) + if (err == nil) != c.ok || (c.ok && !reflect.DeepEqual(got, c.want)) { + t.Errorf("parseCPUList(%q) = %v, %v ; want %v", c.in, got, err, c.want) + } + } +} + +func TestThreadCount(t *testing.T) { + for _, c := range []struct { + in string + n int + note bool + }{ + {"", 0, false}, {"0", 0, false}, {" 8 ", 8, false}, {"auto", 0, true}, {"-1", 0, true}, + } { + n, note := threadCount("THREADS", c.in) + if n != c.n || (note != "") != c.note { + t.Errorf("threadCount(%q) = %d, %q", c.in, n, note) + } + } +} diff --git a/internal/loki/sys_service.go b/internal/loki/sys_service.go index 6ff307d..274e1ac 100644 --- a/internal/loki/sys_service.go +++ b/internal/loki/sys_service.go @@ -70,8 +70,8 @@ var configTemplate = []struct{ key, help string }{ {"NGL", "couches déportées sur le GPU (999 ou auto = ce qui tient en VRAM ; all = tout, quitte à saturer)"}, {"BATCH", "batch (défaut 2048)"}, {"UBATCH", "micro-batch (défaut 512)"}, - {"THREADS", "threads CPU, 0 = auto"}, - {"THREADS_BATCH", "threads CPU du prefill, 0 = auto"}, + {"THREADS", "threads CPU ; vide ou 0 = auto (cœurs physiques ; cœurs P seulement sur Intel hybride Linux)"}, + {"THREADS_BATCH", "threads CPU du prefill et de la vérification spéculative (MTP) ; vide ou 0 = comme THREADS"}, {"KV_TYPE", "quantization du cache KV (q8_0, q4_0…) ; KV_TYPE_K / KV_TYPE_V pour les séparer"}, {"REASONING", "passthrough du mode raisonnement (on/auto/deepseek)"}, {"REASONING_BUDGET", "plafond de tokens de réflexion ; -1 = illimité"}, diff --git a/internal/loki/ui/index.html b/internal/loki/ui/index.html index b3650f4..92c851e 100644 --- a/internal/loki/ui/index.html +++ b/internal/loki/ui/index.html @@ -3441,12 +3441,12 @@ html[data-files="1"] #files-btn{color:var(--accent)}
- Threads CPU0 = auto - + Threads CPUvide ou 0 = cœurs physiques +
- Threads (batch)0 = auto - + Threads (batch)prefill et MTP · vide ou 0 = comme Threads CPU +
BATCH diff --git a/internal/loki/ui/src/index.tmpl.html b/internal/loki/ui/src/index.tmpl.html index 9dd5da6..ef252e0 100644 --- a/internal/loki/ui/src/index.tmpl.html +++ b/internal/loki/ui/src/index.tmpl.html @@ -1218,12 +1218,12 @@ document.documentElement.setAttribute('data-side',localStorage.getItem('loki-sid
- Threads CPU0 = auto - + Threads CPUvide ou 0 = cœurs physiques +
- Threads (batch)0 = auto - + Threads (batch)prefill et MTP · vide ou 0 = comme Threads CPU +
BATCH