fix(chat): heartbeat SSE pendant la préparation + think=False sur les appels auxiliaires

Le déplacement de la préparation (plan, recall RAG) dans le flux SSE
laissait un silence total pendant le chargement d'un gros modèle : le
reverse proxy coupait la connexion (« Connexion interrompue pendant le
chargement du modèle »). Avec la réflexion active, l'appel de plan
laissait en plus le modèle penser — silence encore plus long.

- ping SSE toutes les 10 s pendant la préparation du contexte
- plan / auto-critique / résumé : think=False (repli automatique si le
  modèle refuse le paramètre) — un appel utilitaire ne réfléchit jamais

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
MichaelandClaude Opus 4.8 committed 2026-07-19 01:01:49 +02:00
1 parent da76b8192d
commit a1380d0bf8
3 files changed
+48 -17

No files matched your search

+17 -7
View File
@@ -12,7 +12,7 @@ import re
import httpx
from .ollama_client import ollama
from .ollama_client import OllamaError, ollama
logger = logging.getLogger(__name__)
@@ -61,19 +61,29 @@ async def _ask(
principal : un appel avec des options divergentes force Ollama à recharger
le modèle en plein message.
"""
messages = [{"role": "system", "content": system},
{"role": "user", "content": user}]
opts = {**(options or {}), "temperature": 0.2, "num_predict": num_predict}
# think=False : un appel utilitaire (plan, critique) ne doit jamais
# « réfléchir » — sur un modèle thinking, la pensée dévore le budget et
# multiplie la latence. Repli sans le paramètre si le modèle le refuse.
think: bool | None = False
while True:
text = ""
try:
async for chunk in ollama.chat(
model,
[{"role": "system", "content": system},
{"role": "user", "content": user}],
options={**(options or {}), "temperature": 0.2, "num_predict": num_predict},
keep_alive=keep_alive,
stream=True,
model, messages, options=opts, think=think,
keep_alive=keep_alive, stream=True,
):
text += chunk.get("message", {}).get("content", "")
if chunk.get("done"):
break
return text.strip()
except OllamaError as exc:
if think is False and "think" in str(exc).lower():
think = None
continue
raise
async def make_plan(
+19 -9
View File
@@ -14,7 +14,7 @@ import logging
import httpx
from . import db
from .ollama_client import ollama
from .ollama_client import OllamaError, ollama
logger = logging.getLogger(__name__)
@@ -79,20 +79,30 @@ async def maybe_summarize(
if previous:
transcript = f"[résumé existant] {previous}\n{transcript}"
text = ""
async for chunk in ollama.chat(
model,
[
messages = [
{"role": "system", "content": _SUMMARY_PROMPT},
{"role": "user", "content": transcript},
],
options={**(options or {}), "temperature": 0.2, "num_predict": 350},
keep_alive=keep_alive,
stream=True,
]
opts = {**(options or {}), "temperature": 0.2, "num_predict": 350}
# think=False : le résumé d'arrière-plan ne doit pas « réfléchir »
# (latence ×3 sur un modèle thinking). Repli si paramètre refusé.
think: bool | None = False
while True:
text = ""
try:
async for chunk in ollama.chat(
model, messages, options=opts, think=think,
keep_alive=keep_alive, stream=True,
):
text += chunk.get("message", {}).get("content", "")
if chunk.get("done"):
break
break
except OllamaError as exc:
if think is False and "think" in str(exc).lower():
think = None
continue
raise
if text.strip():
db.set_session_summary(sid, text.strip())
+12 -1
View File
@@ -324,7 +324,10 @@ async def chat(req: ChatRequest) -> StreamingResponse:
from ..mcp_client import manager as mcp_manager
memories, plan, code_model, mcp_tools = await asyncio.gather(
# La préparation peut déclencher un CHARGEMENT de modèle (plan, embed)
# qui dure plusieurs minutes sur un gros modèle. Sans battement de
# cœur, le silence SSE fait couper la connexion par le reverse proxy.
prep = asyncio.ensure_future(asyncio.gather(
rag.recall(req.session_id, req.content, embed_model=cfg.get("embed_model"))
if want_rag else asyncio.sleep(0, result=[]),
enhance.make_plan(model, req.content, options=run_opts, keep_alive=keep)
@@ -332,7 +335,15 @@ async def chat(req: ChatRequest) -> StreamingResponse:
coder.pick_code_model(model, cfg.get("code_model"))
if use_code else asyncio.sleep(0, result=model),
mcp_manager.tool_definitions(),
))
while True:
try:
memories, plan, code_model, mcp_tools = await asyncio.wait_for(
asyncio.shield(prep), timeout=10.0
)
break
except TimeoutError:
yield _sse("ping", {"status": "waiting"})
# Pannes MCP éventuelles : notice non bloquante dans le fil.
for mcp_notice in mcp_manager.notices():