From bb8e87b612706cf4a72224488e5053fff9d7f351 Mon Sep 17 00:00:00 2001 From: Michael SCHAL Date: Sun, 19 Jul 2026 11:28:51 +0200 Subject: [PATCH] =?UTF-8?q?fix(agent):=20coupe-circuit=20de=20r=C3=A9flexi?= =?UTF-8?q?on=20en=20cours=20de=20flux?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit La relance après « pensée seule » n'agissait qu'à la fin de la génération : un modèle thinking pouvait ruminer des minutes (jusqu'à épuiser num_predict) avant qu'elle n'intervienne — vécu comme « aucune réponse » en mode plan. Pensée > 12 000 caractères sans aucun contenu ni appel d'outil : la génération est interrompue immédiatement et la relance repart. Co-Authored-By: Claude Opus 4.8 --- backend/app/agent.py | 14 ++++++++++++++ backend/tests/test_agent_thinking.py | 24 ++++++++++++++++++++++++ 2 files changed, 38 insertions(+) diff --git a/backend/app/agent.py b/backend/app/agent.py index 85fa548..f117c54 100644 --- a/backend/app/agent.py +++ b/backend/app/agent.py @@ -26,6 +26,11 @@ from .tools import TOOL_DEFINITIONS, ToolError, run_tool MAX_ITERATIONS = 6 MAX_TOOL_REPAIR_ATTEMPTS = 2 +# Coupe-circuit de réflexion : au-delà de cette taille de pensée SANS aucun +# contenu ni appel d'outil, on interrompt la génération en cours au lieu +# d'attendre l'épuisement du budget num_predict (minutes sur un gros modèle). +_MAX_THINKING_CHARS = 12_000 + # Élagage : au-delà de cette taille, un résultat d'outil des itérations # passées est tronqué. Les gros payloads (MCP 8 Ko, shell 4 Ko) saturaient le # contexte en un seul tour long — Ollama tronquait alors silencieusement le @@ -185,6 +190,15 @@ async def run_agent( if not thinking_status_sent: thinking_status_sent = True yield {"type": "status", "message": "Réflexion…"} + # Pensée interminable sans production : on coupe la + # génération maintenant — la relance (plus bas) + # remet le modèle au travail immédiatement. + if ( + len(thinking_buf) > _MAX_THINKING_CHARS + and not content_buf + and not tool_calls + ): + break if token: content_buf += token yield {"type": "token", "content": token} diff --git a/backend/tests/test_agent_thinking.py b/backend/tests/test_agent_thinking.py index b7b07ed..7bb7083 100644 --- a/backend/tests/test_agent_thinking.py +++ b/backend/tests/test_agent_thinking.py @@ -75,6 +75,30 @@ async def test_reflexion_pas_renvoyee_au_modele(monkeypatch): assert all("thinking" not in m for m in assistant_turns) +@pytest.mark.asyncio +async def test_coupe_circuit_pensee_interminable(monkeypatch): + """Pensée sans fin -> génération coupée en vol, puis relance qui aboutit.""" + calls: list[int] = [] + + async def fake_chat(model, convo, **kwargs): + calls.append(1) + if len(calls) == 1: + # Flux de pensée « infini » : jamais de done, jamais de contenu. + for _ in range(10_000): + yield {"message": {"thinking": "x" * 200}, "done": False} + else: + yield {"message": {"content": "réponse après coupe"}, "done": True} + + monkeypatch.setattr(agent.ollama, "chat", fake_chat) + events = [ + e async for e in agent.run_agent("test", _convo(), enabled_tools=[]) + ] + + final = [e for e in events if e["type"] == "final"] + assert final and "réponse après coupe" in final[0]["content"] + assert len(calls) == 2 # coupé puis relancé, pas d'épuisement du flux + + @pytest.mark.asyncio async def test_reflexion_coupee_apres_deux_impasses(monkeypatch): """Deux itérations de pensée pure -> think désactivé, tâche finie."""