From fb1294d42c70b949b5c72dd6a85c77a01aa78204 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 18 Aug 2026 10:53:09 +0000 Subject: [PATCH] =?UTF-8?q?Raisonnement=20=C2=AB=20aucune=20=C2=BB=20:=20l?= =?UTF-8?q?e=20dire=20dans=20la=20langue=20du=20gabarit?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Le sélecteur envoyait seulement `reasoning_effort` dans le corps de la requête. Ce champ n'agit que sur les gabarits qui le LISENT (gpt-oss et apparentés). Un modèle hybride à la Qwen3 ne connaît que `enable_thinking` : il recevait `reasoning_effort: none`, son gabarit l'ignorait, et il continuait de réfléchir pendant que l'interface affichait « aucune ». Les llama-server récents traduisent eux-mêmes `none` en `enable_thinking=false`, mais les binaires plus anciens laissent simplement tomber le champ. On joint donc `chat_template_kwargs` à la requête — même procédé que la compaction, qui coupe déjà la réflexion ainsi : aucune (ou REASONING=off explicite) → {"enable_thinking": false} basse/moyenne/haute → {"reasoning_effort": ""} auto → rien, le gabarit garde son défaut Une clé REASONING absente n'est PAS une interdiction : sans consigne on ne touche à rien. Une clé inconnue d'un gabarit est ignorée sans erreur. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01RodUJoPJDBvVKhA6S5msvb --- internal/loki/backend_models.go | 38 +++++++++++++++++++++++++++++++ internal/loki/llm_client.go | 8 +++++++ internal/loki/mcp_catalog_test.go | 37 ++++++++++++++++++++++++++++++ 3 files changed, 83 insertions(+) diff --git a/internal/loki/backend_models.go b/internal/loki/backend_models.go index 08793b3..86d1d51 100644 --- a/internal/loki/backend_models.go +++ b/internal/loki/backend_models.go @@ -98,6 +98,44 @@ func reasoningEffortValue(v string) string { } } +// reasoningTemplateKwargs construit les `chat_template_kwargs` à joindre à une +// requête de chat pour que « aucune » coupe VRAIMENT le raisonnement. +// +// Pourquoi ce doublon avec `reasoning_effort` : ce champ n'agit que sur les +// gabarits qui le LISENT (gpt-oss et apparentés). Un modèle hybride à la Qwen3 +// ne connaît que `enable_thinking` — on lui envoie `reasoning_effort: none`, son +// gabarit l'ignore, et il continue de réfléchir alors que l'interface affiche +// « aucune ». Les llama-server récents traduisent eux-mêmes `none` en +// `enable_thinking=false`, mais pas les binaires plus anciens, qui laissent +// simplement tomber le champ. +// +// D'où la règle : quand le raisonnement doit être coupé, on le dit dans la +// langue que TOUS les gabarits comprennent. `chat_template_kwargs` est passé tel +// quel au gabarit jinja ; une clé qu'il n'utilise pas est ignorée sans erreur. +// Même procédé que la compaction (chat_compact.go), qui coupe déjà la réflexion +// de cette façon. +// +// off = REASONING explicitement à off. Clé absente ≠ off : sans consigne on ne +// touche à rien, le gabarit garde son défaut. +// Renvoie nil quand il n'y a rien à imposer. +func reasoningTemplateKwargs(off bool, effort string) map[string]any { + if off || effort == "none" { + return map[string]any{"enable_thinking": false} + } + if effort != "" { + // Niveau explicite : les vieux binaires ne relaient pas le champ de haut + // niveau au gabarit, celui-ci passe toujours. + return map[string]any{"reasoning_effort": effort} + } + return nil +} + +// reasoningExplicitlyOff dit si REASONING interdit le raisonnement. Une clé +// absente ou vide n'est PAS une interdiction : c'est l'absence de consigne. +func reasoningExplicitlyOff(v string) bool { + return strings.TrimSpace(v) != "" && !reasoningActive(v) +} + // detectQuant returns the quantization tag for a preset: an explicit QUANT= line // (manual override, with or without a leading '#') wins; otherwise it is // auto-detected from the MODEL= filename. Returns "" when unknown. diff --git a/internal/loki/llm_client.go b/internal/loki/llm_client.go index 20b7e4b..3d0b41a 100644 --- a/internal/loki/llm_client.go +++ b/internal/loki/llm_client.go @@ -579,6 +579,11 @@ func runChat(ctx context.Context, messages []Message, temperature float64, caps // Intensité du raisonnement, passée telle quelle au gabarit du modèle. Vide // = on n'envoie rien. Réglage par preset, donc de fait par modèle. reasoningEffort := reasoningEffortValue(chatCfg["REASONING_EFFORT"]) + // Ce que le gabarit du modèle doit savoir, dans SA langue : `reasoning_effort` + // ne parle qu'aux gabarits qui le lisent, `enable_thinking` parle aux modèles + // hybrides (Qwen3 & co). Sans ça, « aucune » n'avait aucun effet sur eux : le + // modèle réfléchissait pendant que l'interface annonçait le contraire. + reasoningKwargs := reasoningTemplateKwargs(reasoningExplicitlyOff(chatCfg["REASONING"]), reasoningEffort) // When llama.cpp fails to parse a model-generated tool call (HTTP 500), we // retry the same turn once with tools removed so the model answers in plain // text from the tool results already gathered, instead of dying mid-chat. @@ -627,6 +632,9 @@ func runChat(ctx context.Context, messages []Message, temperature float64, caps if reasoningEffort != "" { payload["reasoning_effort"] = reasoningEffort } + if reasoningKwargs != nil { + payload["chat_template_kwargs"] = reasoningKwargs + } if len(tools) > 0 && !disableTools { payload["tools"] = tools // The model sometimes emits parallel tool calls, which this llama.cpp diff --git a/internal/loki/mcp_catalog_test.go b/internal/loki/mcp_catalog_test.go index 20c086e..3f88bce 100644 --- a/internal/loki/mcp_catalog_test.go +++ b/internal/loki/mcp_catalog_test.go @@ -79,3 +79,40 @@ func TestReasoningEffortValue(t *testing.T) { } } } + +func TestReasoningExplicitlyOff(t *testing.T) { + // Clé absente ou vide = pas de consigne, surtout pas une interdiction. + for _, v := range []string{"", " ", "on", "auto", "deepseek"} { + if reasoningExplicitlyOff(v) { + t.Errorf("reasoningExplicitlyOff(%q) = true, attendu false", v) + } + } + for _, v := range []string{"off", "none", "0", "false", " Disabled "} { + if !reasoningExplicitlyOff(v) { + t.Errorf("reasoningExplicitlyOff(%q) = false, attendu true", v) + } + } +} + +func TestReasoningTemplateKwargs(t *testing.T) { + // « aucune » doit parler aux gabarits hybrides, pas seulement à ceux qui + // lisent reasoning_effort : sinon le modèle réfléchit quand même. + if got := reasoningTemplateKwargs(false, "none"); got == nil || got["enable_thinking"] != false { + t.Errorf("effort none : %v, attendu enable_thinking=false", got) + } + if got := reasoningTemplateKwargs(true, ""); got == nil || got["enable_thinking"] != false { + t.Errorf("REASONING=off : %v, attendu enable_thinking=false", got) + } + // Un niveau explicite voyage aussi dans les kwargs : les binaires anciens ne + // relaient pas le champ de haut niveau au gabarit. + if got := reasoningTemplateKwargs(false, "high"); got == nil || got["reasoning_effort"] != "high" { + t.Errorf("effort high : %v, attendu reasoning_effort=high", got) + } + if _, ok := reasoningTemplateKwargs(false, "high")["enable_thinking"]; ok { + t.Error("un niveau explicite ne doit pas couper le raisonnement") + } + // Auto = aucune consigne : on n'écrit rien dans la requête. + if got := reasoningTemplateKwargs(false, ""); got != nil { + t.Errorf("auto : %v, attendu nil", got) + } +}