From 5c1b92daa3ce6d342ab65404e91908456d966ae5 Mon Sep 17 00:00:00 2001 From: R0m1k3 Date: Tue, 30 Jun 2026 14:29:18 +0200 Subject: [PATCH] =?UTF-8?q?Ajoute=20un=20interrupteur=20=C2=AB=20Mode=20r?= =?UTF-8?q?=C3=A9flexion=20=C2=BB=20(think)=20pour=20les=20mod=C3=A8les=20?= =?UTF-8?q?thinking?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Certains modèles « thinking » ne renvoyaient que du raisonnement, sans reponse finale -> erreur. Nouvel interrupteur global `think` (Reglages > Generation) : - think=false envoie `think:false` a Ollama pour desactiver le raisonnement - repli automatique si le modele refuse le parametre (does not support thinking) - message d'erreur mis a jour pour pointer vers l'interrupteur Co-Authored-By: Claude Opus 4.8 --- backend/app/agent.py | 24 +++++++++++++++++++++--- backend/app/agent_config.py | 3 +++ backend/app/ollama_client.py | 4 ++++ backend/app/routes/chat.py | 1 + backend/app/routes/config.py | 1 + frontend/src/api/client.ts | 1 + frontend/src/panels/SettingsView.tsx | 10 ++++++++++ 7 files changed, 41 insertions(+), 3 deletions(-) diff --git a/backend/app/agent.py b/backend/app/agent.py index a2d1ccb..d03d4f2 100644 --- a/backend/app/agent.py +++ b/backend/app/agent.py @@ -44,6 +44,12 @@ def _tools_not_supported(exc: OllamaError) -> bool: ) +def _thinking_not_supported(exc: OllamaError) -> bool: + """Le modèle (ou son template) refuse le paramètre ``think``.""" + message = str(exc).lower() + return "does not support thinking" in message or "thinking is not supported" in message + + def _invalid_tool_arguments(exc: OllamaError) -> bool: message = str(exc).lower() return ( @@ -71,6 +77,7 @@ async def run_agent( options: dict | None = None, enabled_tools: list[str] | None = None, confirm_shell: bool = True, + think: bool = True, ) -> AsyncIterator[dict]: # enabled_tools=None -> tous les outils ; liste vide -> aucun outil. if enabled_tools is None: @@ -88,6 +95,9 @@ async def run_agent( tool_fallback_used = False tool_repair_attempts = 0 request_options = dict(options or {}) + # On n'envoie ``think`` que pour le DÉSACTIVER (False) ; laissé à None, le + # modèle garde son comportement par défaut. Repli si le modèle le refuse. + request_think: bool | None = None if think else False # Métriques cumulées sur tous les appels Ollama du tour agentique : Ollama # les renvoie dans le chunk final (done=true) de chaque génération. @@ -114,6 +124,7 @@ async def run_agent( convo, tools=active_tools, options=request_options, + think=request_think, stream=True, ): msg = chunk.get("message", {}) @@ -191,6 +202,11 @@ async def run_agent( ), } continue + if request_think is not None and _thinking_not_supported(exc): + # Le modèle n'accepte pas qu'on désactive sa réflexion : + # on retire le paramètre et on relance. + request_think = None + continue raise assistant_turn: dict = {"role": "assistant", "content": content_buf} @@ -270,7 +286,8 @@ async def run_agent( # Laisse le modèle conclure son tour (message d'attente). final_chunk = "" async for chunk in ollama.chat( - model, convo, options=request_options, stream=True + model, convo, options=request_options, + think=request_think, stream=True ): tok = chunk.get("message", {}).get("content", "") if tok: @@ -299,8 +316,9 @@ async def run_agent( yield { "type": "error", "message": ( - "Le modèle a terminé sans renvoyer de texte. Essayez un modèle " - "de chat récent ou désactivez son mode de réflexion avancée." + "Le modèle a terminé sans renvoyer de texte (il n'a produit que " + "du raisonnement). Désactive « Mode réflexion » dans les Réglages, " + "ou essaie un modèle de chat plus récent." ), } return diff --git a/backend/app/agent_config.py b/backend/app/agent_config.py index da1c2d0..e19f46a 100644 --- a/backend/app/agent_config.py +++ b/backend/app/agent_config.py @@ -72,6 +72,9 @@ DEFAULT_CONFIG: dict = { "tools": dict(DEFAULT_TOOL_STATE), # Demander une validation utilisateur avant toute commande shell. "confirm_shell": True, + # Mode réflexion des modèles « thinking ». Désactiver (False) évite qu'un + # modèle ne renvoie que du raisonnement sans réponse finale. + "think": True, } diff --git a/backend/app/ollama_client.py b/backend/app/ollama_client.py index 533f1cb..37b396c 100644 --- a/backend/app/ollama_client.py +++ b/backend/app/ollama_client.py @@ -106,6 +106,7 @@ class OllamaClient: *, tools: list[dict] | None = None, options: dict | None = None, + think: bool | None = None, stream: bool = True, ) -> AsyncIterator[dict]: """Conversation avec le modèle, en streaming token par token.""" @@ -114,6 +115,9 @@ class OllamaClient: payload["tools"] = tools if options: payload["options"] = options + # think=False désactive le raisonnement des modèles « thinking ». + if think is not None: + payload["think"] = think async with httpx.AsyncClient( timeout=_STREAM_TIMEOUT, follow_redirects=True diff --git a/backend/app/routes/chat.py b/backend/app/routes/chat.py index 684a43b..af245cf 100644 --- a/backend/app/routes/chat.py +++ b/backend/app/routes/chat.py @@ -72,6 +72,7 @@ async def chat(req: ChatRequest) -> StreamingResponse: options=agent_config.ollama_options(cfg), enabled_tools=agent_config.enabled_tool_names(cfg), confirm_shell=cfg.get("confirm_shell", True), + think=cfg.get("think", True), ): await queue.put(event) except Exception as exc: diff --git a/backend/app/routes/config.py b/backend/app/routes/config.py index 8dc7e85..3c9e999 100644 --- a/backend/app/routes/config.py +++ b/backend/app/routes/config.py @@ -20,6 +20,7 @@ class ConfigPatch(BaseModel): num_batch: int | None = None tools: dict[str, bool] | None = None confirm_shell: bool | None = None + think: bool | None = None @router.get("") diff --git a/frontend/src/api/client.ts b/frontend/src/api/client.ts index a7e1b37..e08a221 100644 --- a/frontend/src/api/client.ts +++ b/frontend/src/api/client.ts @@ -72,6 +72,7 @@ export interface AgentConfig { num_batch: number; tools: Record; confirm_shell: boolean; + think: boolean; } export async function runShell( diff --git a/frontend/src/panels/SettingsView.tsx b/frontend/src/panels/SettingsView.tsx index e3b1d53..baf7db7 100644 --- a/frontend/src/panels/SettingsView.tsx +++ b/frontend/src/panels/SettingsView.tsx @@ -289,6 +289,16 @@ export function SettingsView() { onChange={(v) => set("num_batch", Math.round(v))} last /> +
+
+
Mode réflexion
+
+ Désactive si le modèle ne renvoie que du raisonnement sans + réponse. +
+
+ set("think", !draft.think)} /> +
La précision KV (f16/q8_0) est un réglage global du serveur Ollama et nécessite son redémarrage.