diff --git a/internal/loki/backend_models.go b/internal/loki/backend_models.go index 08793b3..86d1d51 100644 --- a/internal/loki/backend_models.go +++ b/internal/loki/backend_models.go @@ -98,6 +98,44 @@ func reasoningEffortValue(v string) string { } } +// reasoningTemplateKwargs construit les `chat_template_kwargs` à joindre à une +// requête de chat pour que « aucune » coupe VRAIMENT le raisonnement. +// +// Pourquoi ce doublon avec `reasoning_effort` : ce champ n'agit que sur les +// gabarits qui le LISENT (gpt-oss et apparentés). Un modèle hybride à la Qwen3 +// ne connaît que `enable_thinking` — on lui envoie `reasoning_effort: none`, son +// gabarit l'ignore, et il continue de réfléchir alors que l'interface affiche +// « aucune ». Les llama-server récents traduisent eux-mêmes `none` en +// `enable_thinking=false`, mais pas les binaires plus anciens, qui laissent +// simplement tomber le champ. +// +// D'où la règle : quand le raisonnement doit être coupé, on le dit dans la +// langue que TOUS les gabarits comprennent. `chat_template_kwargs` est passé tel +// quel au gabarit jinja ; une clé qu'il n'utilise pas est ignorée sans erreur. +// Même procédé que la compaction (chat_compact.go), qui coupe déjà la réflexion +// de cette façon. +// +// off = REASONING explicitement à off. Clé absente ≠ off : sans consigne on ne +// touche à rien, le gabarit garde son défaut. +// Renvoie nil quand il n'y a rien à imposer. +func reasoningTemplateKwargs(off bool, effort string) map[string]any { + if off || effort == "none" { + return map[string]any{"enable_thinking": false} + } + if effort != "" { + // Niveau explicite : les vieux binaires ne relaient pas le champ de haut + // niveau au gabarit, celui-ci passe toujours. + return map[string]any{"reasoning_effort": effort} + } + return nil +} + +// reasoningExplicitlyOff dit si REASONING interdit le raisonnement. Une clé +// absente ou vide n'est PAS une interdiction : c'est l'absence de consigne. +func reasoningExplicitlyOff(v string) bool { + return strings.TrimSpace(v) != "" && !reasoningActive(v) +} + // detectQuant returns the quantization tag for a preset: an explicit QUANT= line // (manual override, with or without a leading '#') wins; otherwise it is // auto-detected from the MODEL= filename. Returns "" when unknown. diff --git a/internal/loki/llm_client.go b/internal/loki/llm_client.go index 20b7e4b..3d0b41a 100644 --- a/internal/loki/llm_client.go +++ b/internal/loki/llm_client.go @@ -579,6 +579,11 @@ func runChat(ctx context.Context, messages []Message, temperature float64, caps // Intensité du raisonnement, passée telle quelle au gabarit du modèle. Vide // = on n'envoie rien. Réglage par preset, donc de fait par modèle. reasoningEffort := reasoningEffortValue(chatCfg["REASONING_EFFORT"]) + // Ce que le gabarit du modèle doit savoir, dans SA langue : `reasoning_effort` + // ne parle qu'aux gabarits qui le lisent, `enable_thinking` parle aux modèles + // hybrides (Qwen3 & co). Sans ça, « aucune » n'avait aucun effet sur eux : le + // modèle réfléchissait pendant que l'interface annonçait le contraire. + reasoningKwargs := reasoningTemplateKwargs(reasoningExplicitlyOff(chatCfg["REASONING"]), reasoningEffort) // When llama.cpp fails to parse a model-generated tool call (HTTP 500), we // retry the same turn once with tools removed so the model answers in plain // text from the tool results already gathered, instead of dying mid-chat. @@ -627,6 +632,9 @@ func runChat(ctx context.Context, messages []Message, temperature float64, caps if reasoningEffort != "" { payload["reasoning_effort"] = reasoningEffort } + if reasoningKwargs != nil { + payload["chat_template_kwargs"] = reasoningKwargs + } if len(tools) > 0 && !disableTools { payload["tools"] = tools // The model sometimes emits parallel tool calls, which this llama.cpp diff --git a/internal/loki/mcp_catalog_test.go b/internal/loki/mcp_catalog_test.go index 20c086e..3f88bce 100644 --- a/internal/loki/mcp_catalog_test.go +++ b/internal/loki/mcp_catalog_test.go @@ -79,3 +79,40 @@ func TestReasoningEffortValue(t *testing.T) { } } } + +func TestReasoningExplicitlyOff(t *testing.T) { + // Clé absente ou vide = pas de consigne, surtout pas une interdiction. + for _, v := range []string{"", " ", "on", "auto", "deepseek"} { + if reasoningExplicitlyOff(v) { + t.Errorf("reasoningExplicitlyOff(%q) = true, attendu false", v) + } + } + for _, v := range []string{"off", "none", "0", "false", " Disabled "} { + if !reasoningExplicitlyOff(v) { + t.Errorf("reasoningExplicitlyOff(%q) = false, attendu true", v) + } + } +} + +func TestReasoningTemplateKwargs(t *testing.T) { + // « aucune » doit parler aux gabarits hybrides, pas seulement à ceux qui + // lisent reasoning_effort : sinon le modèle réfléchit quand même. + if got := reasoningTemplateKwargs(false, "none"); got == nil || got["enable_thinking"] != false { + t.Errorf("effort none : %v, attendu enable_thinking=false", got) + } + if got := reasoningTemplateKwargs(true, ""); got == nil || got["enable_thinking"] != false { + t.Errorf("REASONING=off : %v, attendu enable_thinking=false", got) + } + // Un niveau explicite voyage aussi dans les kwargs : les binaires anciens ne + // relaient pas le champ de haut niveau au gabarit. + if got := reasoningTemplateKwargs(false, "high"); got == nil || got["reasoning_effort"] != "high" { + t.Errorf("effort high : %v, attendu reasoning_effort=high", got) + } + if _, ok := reasoningTemplateKwargs(false, "high")["enable_thinking"]; ok { + t.Error("un niveau explicite ne doit pas couper le raisonnement") + } + // Auto = aucune consigne : on n'écrit rien dans la requête. + if got := reasoningTemplateKwargs(false, ""); got != nil { + t.Errorf("auto : %v, attendu nil", got) + } +}