mirror of
https://github.com/R0m1k3/Loki.git
synced 2026-10-11 17:26:57 +02:00
Avec la réflexion activée sur les longues tâches, le modèle brûlait tout son budget de génération à penser : done sans contenu ni appel d'outil, la boucle s'arrêtait et la tâche restait inachevée. - itération « réflexion seule » : relance avec consigne d'agir ; si ça recommence, think est coupé pour la fin du tour (+ notice) - marge num_predict >= 6144 quand la réflexion est active - la pensée reste affichée dans l'UI mais n'est plus renvoyée au modèle (elle regonflait le contexte à chaque itération) Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
101 lines
3.5 KiB
Python
101 lines
3.5 KiB
Python
import os
|
|
import tempfile
|
|
|
|
os.environ.setdefault("DATA_DIR", tempfile.mkdtemp())
|
|
os.environ.setdefault("WORKSPACE_DIR", tempfile.mkdtemp())
|
|
|
|
import pytest # noqa: E402
|
|
|
|
from app import agent # noqa: E402
|
|
|
|
|
|
def _convo():
|
|
return [
|
|
{"role": "system", "content": "consigne"},
|
|
{"role": "user", "content": "fais la tâche"},
|
|
]
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_relance_apres_reflexion_seule(monkeypatch):
|
|
"""1er appel : réflexion seule -> relance ; 2e appel : réponse finale."""
|
|
calls: list[list[dict]] = []
|
|
|
|
async def fake_chat(model, convo, **kwargs):
|
|
calls.append([dict(m) for m in convo])
|
|
if len(calls) == 1:
|
|
yield {"message": {"thinking": "hmm, je réfléchis longuement…"},
|
|
"done": False}
|
|
yield {"message": {}, "done": True}
|
|
else:
|
|
yield {"message": {"content": "réponse finale"}, "done": True}
|
|
|
|
monkeypatch.setattr(agent.ollama, "chat", fake_chat)
|
|
events = [
|
|
e async for e in agent.run_agent("test", _convo(), enabled_tools=[])
|
|
]
|
|
|
|
final = [e for e in events if e["type"] == "final"]
|
|
assert final and "réponse finale" in final[0]["content"]
|
|
# La relance a bien injecté la consigne de reprise.
|
|
assert any(
|
|
"Continue la tâche" in m["content"]
|
|
for m in calls[1] if m["role"] == "user"
|
|
)
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_reflexion_pas_renvoyee_au_modele(monkeypatch):
|
|
"""La pensée est affichée mais jamais réinjectée dans l'historique."""
|
|
calls: list[list[dict]] = []
|
|
|
|
async def fake_chat(model, convo, **kwargs):
|
|
calls.append([dict(m) for m in convo])
|
|
if len(calls) == 1:
|
|
yield {"message": {"thinking": "je planifie",
|
|
"content": "étape 1",
|
|
"tool_calls": [{"function": {
|
|
"name": "list_dir",
|
|
"arguments": {"path": "."}}}]},
|
|
"done": True}
|
|
else:
|
|
yield {"message": {"content": "terminé"}, "done": True}
|
|
|
|
monkeypatch.setattr(agent.ollama, "chat", fake_chat)
|
|
events = [
|
|
e async for e in agent.run_agent("test", _convo(), enabled_tools=None)
|
|
]
|
|
|
|
final = [e for e in events if e["type"] == "final"]
|
|
assert final and "terminé" in final[0]["content"]
|
|
assert "je planifie" in final[0]["thinking"] # gardée pour l'UI
|
|
# Aucun message assistant réinjecté ne contient la clé thinking.
|
|
assistant_turns = [m for m in calls[1] if m["role"] == "assistant"]
|
|
assert assistant_turns
|
|
assert all("thinking" not in m for m in assistant_turns)
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_reflexion_coupee_apres_deux_impasses(monkeypatch):
|
|
"""Deux itérations de pensée pure -> think désactivé, tâche finie."""
|
|
seen_think: list = []
|
|
|
|
async def fake_chat(model, convo, think=None, **kwargs):
|
|
seen_think.append(think)
|
|
if len(seen_think) <= 2:
|
|
yield {"message": {"thinking": "boucle de pensée"}, "done": True}
|
|
else:
|
|
yield {"message": {"content": "fini sans réfléchir"}, "done": True}
|
|
|
|
monkeypatch.setattr(agent.ollama, "chat", fake_chat)
|
|
events = [
|
|
e async for e in agent.run_agent(
|
|
"test", _convo(), enabled_tools=[], think=True
|
|
)
|
|
]
|
|
|
|
final = [e for e in events if e["type"] == "final"]
|
|
assert final and "fini sans réfléchir" in final[0]["content"]
|
|
assert seen_think[-1] is False # think coupé pour l'appel final
|
|
assert any(e["type"] == "notice" for e in events)
|