From da76b8192d29a5f0dfcaa98a237b1694a9125805 Mon Sep 17 00:00:00 2001 From: Michael SCHAL Date: Sun, 19 Jul 2026 00:48:23 +0200 Subject: [PATCH] =?UTF-8?q?fix(agent):=20mode=20r=C3=A9flexion=20=E2=80=94?= =?UTF-8?q?=20relance=20sur=20pens=C3=A9e=20pure=20+=20pens=C3=A9e=20hors?= =?UTF-8?q?=20historique?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Avec la réflexion activée sur les longues tâches, le modèle brûlait tout son budget de génération à penser : done sans contenu ni appel d'outil, la boucle s'arrêtait et la tâche restait inachevée. - itération « réflexion seule » : relance avec consigne d'agir ; si ça recommence, think est coupé pour la fin du tour (+ notice) - marge num_predict >= 6144 quand la réflexion est active - la pensée reste affichée dans l'UI mais n'est plus renvoyée au modèle (elle regonflait le contexte à chaque itération) Co-Authored-By: Claude Opus 4.8 --- backend/app/agent.py | 41 ++++++++++- backend/tests/test_agent_thinking.py | 100 +++++++++++++++++++++++++++ 2 files changed, 140 insertions(+), 1 deletion(-) create mode 100644 backend/tests/test_agent_thinking.py diff --git a/backend/app/agent.py b/backend/app/agent.py index cb41859..11191f0 100644 --- a/backend/app/agent.py +++ b/backend/app/agent.py @@ -128,6 +128,14 @@ async def run_agent( # On n'envoie ``think`` que pour le DÉSACTIVER (False) ; laissé à None, le # modèle garde son comportement par défaut. Repli si le modèle le refuse. request_think: bool | None = None if think else False + # Mode réflexion : la pensée consomme le budget num_predict. Sans marge, + # le modèle « pense » tout son quota et s'arrête sans agir ni répondre. + if request_think is None and think: + request_options["num_predict"] = max( + int(request_options.get("num_predict") or 0), 6144 + ) + # Compteur d'itérations où le modèle n'a produit QUE du raisonnement. + thinking_only_strikes = 0 # Métriques cumulées sur tous les appels Ollama du tour agentique : Ollama # les renvoie dans le chunk final (done=true) de chaque génération. @@ -248,9 +256,40 @@ async def run_agent( continue raise + # Itération « réflexion seule » : ni contenu, ni appel d'outil — + # le modèle a brûlé sa génération à penser. Sans relance, la + # boucle s'arrêtait là et la tâche restait inachevée. + if thinking_buf and not content_buf.strip() and not tool_calls: + thinking_parts.append(thinking_buf) + thinking_only_strikes += 1 + if thinking_only_strikes == 1: + convo.append({ + "role": "user", + "content": ( + "Tu n'as produit que du raisonnement, sans réponse " + "ni action. Continue la tâche MAINTENANT : appelle " + "l'outil suivant ou donne ta réponse finale, sans " + "réfléchir davantage." + ), + }) + yield {"type": "status", "message": "Relance après réflexion…"} + continue + # Deuxième fois : la réflexion est coupée pour finir la tâche. + request_think = False + yield { + "type": "notice", + "message": ( + "Réflexion désactivée pour ce tour : le modèle " + "n'avançait plus." + ), + } + continue + assistant_turn: dict = {"role": "assistant", "content": content_buf} if thinking_buf: - assistant_turn["thinking"] = thinking_buf + # Gardée pour l'affichage (panneau repliable), mais JAMAIS + # renvoyée au modèle : la re-soumettre gonflait le contexte à + # chaque itération des longues tâches. thinking_parts.append(thinking_buf) if tool_calls: assistant_turn["tool_calls"] = tool_calls diff --git a/backend/tests/test_agent_thinking.py b/backend/tests/test_agent_thinking.py new file mode 100644 index 0000000..b7b07ed --- /dev/null +++ b/backend/tests/test_agent_thinking.py @@ -0,0 +1,100 @@ +import os +import tempfile + +os.environ.setdefault("DATA_DIR", tempfile.mkdtemp()) +os.environ.setdefault("WORKSPACE_DIR", tempfile.mkdtemp()) + +import pytest # noqa: E402 + +from app import agent # noqa: E402 + + +def _convo(): + return [ + {"role": "system", "content": "consigne"}, + {"role": "user", "content": "fais la tâche"}, + ] + + +@pytest.mark.asyncio +async def test_relance_apres_reflexion_seule(monkeypatch): + """1er appel : réflexion seule -> relance ; 2e appel : réponse finale.""" + calls: list[list[dict]] = [] + + async def fake_chat(model, convo, **kwargs): + calls.append([dict(m) for m in convo]) + if len(calls) == 1: + yield {"message": {"thinking": "hmm, je réfléchis longuement…"}, + "done": False} + yield {"message": {}, "done": True} + else: + yield {"message": {"content": "réponse finale"}, "done": True} + + monkeypatch.setattr(agent.ollama, "chat", fake_chat) + events = [ + e async for e in agent.run_agent("test", _convo(), enabled_tools=[]) + ] + + final = [e for e in events if e["type"] == "final"] + assert final and "réponse finale" in final[0]["content"] + # La relance a bien injecté la consigne de reprise. + assert any( + "Continue la tâche" in m["content"] + for m in calls[1] if m["role"] == "user" + ) + + +@pytest.mark.asyncio +async def test_reflexion_pas_renvoyee_au_modele(monkeypatch): + """La pensée est affichée mais jamais réinjectée dans l'historique.""" + calls: list[list[dict]] = [] + + async def fake_chat(model, convo, **kwargs): + calls.append([dict(m) for m in convo]) + if len(calls) == 1: + yield {"message": {"thinking": "je planifie", + "content": "étape 1", + "tool_calls": [{"function": { + "name": "list_dir", + "arguments": {"path": "."}}}]}, + "done": True} + else: + yield {"message": {"content": "terminé"}, "done": True} + + monkeypatch.setattr(agent.ollama, "chat", fake_chat) + events = [ + e async for e in agent.run_agent("test", _convo(), enabled_tools=None) + ] + + final = [e for e in events if e["type"] == "final"] + assert final and "terminé" in final[0]["content"] + assert "je planifie" in final[0]["thinking"] # gardée pour l'UI + # Aucun message assistant réinjecté ne contient la clé thinking. + assistant_turns = [m for m in calls[1] if m["role"] == "assistant"] + assert assistant_turns + assert all("thinking" not in m for m in assistant_turns) + + +@pytest.mark.asyncio +async def test_reflexion_coupee_apres_deux_impasses(monkeypatch): + """Deux itérations de pensée pure -> think désactivé, tâche finie.""" + seen_think: list = [] + + async def fake_chat(model, convo, think=None, **kwargs): + seen_think.append(think) + if len(seen_think) <= 2: + yield {"message": {"thinking": "boucle de pensée"}, "done": True} + else: + yield {"message": {"content": "fini sans réfléchir"}, "done": True} + + monkeypatch.setattr(agent.ollama, "chat", fake_chat) + events = [ + e async for e in agent.run_agent( + "test", _convo(), enabled_tools=[], think=True + ) + ] + + final = [e for e in events if e["type"] == "final"] + assert final and "fini sans réfléchir" in final[0]["content"] + assert seen_think[-1] is False # think coupé pour l'appel final + assert any(e["type"] == "notice" for e in events)