mirror of
https://github.com/R0m1k3/Loki.git
synced 2026-10-11 17:26:57 +02:00
Moteur : corrections de relecture des threads CPU — un -tb absent n'efface plus les réglages batch d'EXTRA_ARGS
Sans -tb, llama.cpp ne recopie pas seulement le nombre de -t : il remplace TOUT le réglage CPU du batch par celui de -t (postprocess_cpu_params, « cpuparams = *role_model »). Un -Cb, -Crb, --cpu-strict-batch, --prio-batch ou --poll-batch écrit dans EXTRA_ARGS était donc effacé sans un mot depuis que Loki ne pose plus « -tb 0 » — un réglage utilisateur cassé. - Dans ce cas seulement, Loki pose -tb : la valeur de -t quand elle est connue (THREADS, sonde du conteneur ou -t / --threads d'EXTRA_ARGS), sinon 0, l'ancien comportement, et le dit sur stderr. - Un -tb déjà dans EXTRA_ARGS reste maître, comme avant. - flagValue lit la dernière occurrence d'un drapeau (« -t 6 » ou « -t=6 »). - Tests : quatre lignes de construction d'arguments et une table flagValue. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
1 parent
c6c5dc2b7e
commit
0a02604968
3 files changed
+92
-3
No files matched your search
@@ -117,6 +117,32 @@ func TestBuildServeArgs(t *testing.T) {
|
||||
si: serveSysInfo{Help: helpRecent},
|
||||
want: with("--parallel", "1", "-ngl", "28", "-t=6", "--threads-batch", "10"),
|
||||
},
|
||||
{
|
||||
// Sans -tb, llama.cpp remplacerait tout le réglage CPU du batch par
|
||||
// celui de -t : le --poll-batch de l'utilisateur serait effacé.
|
||||
name: "--poll-batch dans EXTRA_ARGS, THREADS vide : -tb 0 le protège, et on le dit",
|
||||
cfg: map[string]string{"NGL": "28", "EXTRA_ARGS": "--poll-batch 0"},
|
||||
si: serveSysInfo{Help: helpRecent},
|
||||
want: withThreads([]string{"-tb", "0"}, "--parallel", "1", "-ngl", "28", "--poll-batch", "0"), wantNotes: 1,
|
||||
},
|
||||
{
|
||||
name: "-Cb dans EXTRA_ARGS, THREADS=6 : -tb recopie 6",
|
||||
cfg: map[string]string{"NGL": "28", "THREADS": "6", "EXTRA_ARGS": "-Cb 0xff"},
|
||||
si: serveSysInfo{Help: helpRecent},
|
||||
want: withThreads([]string{"-t", "6", "-tb", "6"}, "--parallel", "1", "-ngl", "28", "-Cb", "0xff"),
|
||||
},
|
||||
{
|
||||
name: "-t=5 et --prio-batch dans EXTRA_ARGS : -tb recopie 5",
|
||||
cfg: map[string]string{"NGL": "28", "EXTRA_ARGS": "-t=5 --prio-batch 2"},
|
||||
si: serveSysInfo{Help: helpRecent},
|
||||
want: withThreads([]string{"-tb", "5"}, "--parallel", "1", "-ngl", "28", "-t=5", "--prio-batch", "2"),
|
||||
},
|
||||
{
|
||||
name: "-Crb et -tb dans EXTRA_ARGS : Loki se tait",
|
||||
cfg: map[string]string{"NGL": "28", "EXTRA_ARGS": "-Crb 0-7 -tb 8"},
|
||||
si: serveSysInfo{Help: helpRecent},
|
||||
want: with("--parallel", "1", "-ngl", "28", "-Crb", "0-7", "-tb", "8"),
|
||||
},
|
||||
{
|
||||
name: "conteneur à l'étroit : -t de la sonde, et une note",
|
||||
cfg: map[string]string{"NGL": "28"},
|
||||
|
||||
@@ -65,21 +65,33 @@ func threadCount(key, v string) (int, string) {
|
||||
// THREADS vide/0 → rien (cœurs physiques, choisis par llama.cpp)
|
||||
// THREADS_BATCH=N>0 → -tb N
|
||||
// THREADS_BATCH vide/0 → rien (le moteur recopie -t)
|
||||
// … sauf -Cb/--poll-batch… → -tb <valeur de -t, sinon 0> (voir plus bas)
|
||||
//
|
||||
// Un -t / -tb déjà écrit dans EXTRA_ARGS gagne : on ne double pas le drapeau.
|
||||
// Un masque d'affinité posé à la main (-C, -Cr, --cpu-strict) désarme la sonde :
|
||||
// l'utilisateur a pris la main sur le placement, on ne devine rien par-dessus.
|
||||
//
|
||||
// Piège du -tb absent : llama.cpp ne recopie pas seulement le NOMBRE de -t, il
|
||||
// remplace TOUT le réglage CPU du batch par celui de -t (postprocess_cpu_params,
|
||||
// « cpuparams = *role_model »). Un -Cb, --poll-batch ou --prio-batch écrit dans
|
||||
// EXTRA_ARGS serait alors effacé sans un mot — le -tb 0 d'avant le protégeait.
|
||||
// Dans ce cas seulement, on pose donc -tb : la valeur de -t si on la connaît
|
||||
// (« comme THREADS »), sinon 0, l'ancien comportement.
|
||||
func threadArgs(threads, threadsBatch string, extra []string, auto cpuBudget) (args, notes []string) {
|
||||
t, note := threadCount("THREADS", threads)
|
||||
if note != "" {
|
||||
notes = append(notes, note)
|
||||
}
|
||||
tVal := "" // valeur de -t effective, si on la connaît
|
||||
switch {
|
||||
case hasAnyFlag(extra, "-t", "--threads"):
|
||||
tVal = flagValue(extra, "-t", "--threads")
|
||||
case t > 0:
|
||||
args = append(args, "-t", strconv.Itoa(t))
|
||||
tVal = strconv.Itoa(t)
|
||||
args = append(args, "-t", tVal)
|
||||
case auto.N > 0 && !hasAnyFlag(extra, "-C", "--cpu-mask", "-Cr", "--cpu-range", "--cpu-strict"):
|
||||
args = append(args, "-t", strconv.Itoa(auto.N))
|
||||
tVal = strconv.Itoa(auto.N)
|
||||
args = append(args, "-t", tVal)
|
||||
notes = append(notes, fmt.Sprintf("threads auto → -t %d : %s (llama.cpp seul en lancerait %d)",
|
||||
auto.N, auto.Why, auto.Engine))
|
||||
}
|
||||
@@ -87,12 +99,45 @@ func threadArgs(threads, threadsBatch string, extra []string, auto cpuBudget) (a
|
||||
if note != "" {
|
||||
notes = append(notes, note)
|
||||
}
|
||||
if tb > 0 && !hasAnyFlag(extra, "-tb", "--threads-batch") {
|
||||
switch {
|
||||
case hasAnyFlag(extra, "-tb", "--threads-batch"):
|
||||
case tb > 0:
|
||||
args = append(args, "-tb", strconv.Itoa(tb))
|
||||
case hasAnyFlag(extra, batchCPUFlags...):
|
||||
if tVal == "" {
|
||||
tVal = "0"
|
||||
notes = append(notes, "EXTRA_ARGS règle le CPU du batch (-Cb, --poll-batch…) : -tb 0 posé pour que "+
|
||||
"llama.cpp le garde (tous les threads logiques) ; THREADS_BATCH pour choisir le nombre")
|
||||
}
|
||||
args = append(args, "-tb", tVal)
|
||||
}
|
||||
return args, notes
|
||||
}
|
||||
|
||||
// batchCPUFlags : les réglages CPU propres au batch, qu'un -tb absent effacerait.
|
||||
var batchCPUFlags = []string{"-Cb", "--cpu-mask-batch", "-Crb", "--cpu-range-batch",
|
||||
"--cpu-strict-batch", "--prio-batch", "--poll-batch"}
|
||||
|
||||
// flagValue renvoie la valeur de la DERNIÈRE occurrence de l'un des drapeaux
|
||||
// (« -t 6 » ou « -t=6 ») — celle que retient le moteur. Vide s'il n'y en a pas.
|
||||
func flagValue(args []string, flags ...string) string {
|
||||
v := ""
|
||||
for i, a := range args {
|
||||
name, val, hasEq := strings.Cut(a, "=")
|
||||
for _, f := range flags {
|
||||
if name != f {
|
||||
continue
|
||||
}
|
||||
if hasEq {
|
||||
v = val
|
||||
} else if i+1 < len(args) {
|
||||
v = args[i+1]
|
||||
}
|
||||
}
|
||||
}
|
||||
return v
|
||||
}
|
||||
|
||||
// cpusetThreads : sonde de conteneur, Linux seulement (fsys = la racine « / »).
|
||||
//
|
||||
// L'auto de llama.cpp se trompe quand le conteneur n'a droit qu'à une partie
|
||||
|
||||
@@ -143,3 +143,21 @@ func TestThreadCount(t *testing.T) {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestFlagValue(t *testing.T) {
|
||||
for _, c := range []struct {
|
||||
args []string
|
||||
want string
|
||||
}{
|
||||
{nil, ""},
|
||||
{[]string{"-t", "6"}, "6"},
|
||||
{[]string{"--threads=7"}, "7"},
|
||||
{[]string{"-t", "6", "--threads", "8"}, "8"}, // la dernière gagne, comme dans le moteur
|
||||
{[]string{"-tb", "9"}, ""}, // -tb n'est pas -t
|
||||
{[]string{"-t"}, ""}, // drapeau orphelin
|
||||
} {
|
||||
if got := flagValue(c.args, "-t", "--threads"); got != c.want {
|
||||
t.Errorf("flagValue(%q) = %q, want %q", c.args, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
Reference in new issue
Block a user