diff --git a/backend/app/enhance.py b/backend/app/enhance.py index 921051d..6bd9737 100644 --- a/backend/app/enhance.py +++ b/backend/app/enhance.py @@ -12,7 +12,7 @@ import re import httpx -from .ollama_client import ollama +from .ollama_client import OllamaError, ollama logger = logging.getLogger(__name__) @@ -61,19 +61,29 @@ async def _ask( principal : un appel avec des options divergentes force Ollama à recharger le modèle en plein message. """ - text = "" - async for chunk in ollama.chat( - model, - [{"role": "system", "content": system}, - {"role": "user", "content": user}], - options={**(options or {}), "temperature": 0.2, "num_predict": num_predict}, - keep_alive=keep_alive, - stream=True, - ): - text += chunk.get("message", {}).get("content", "") - if chunk.get("done"): - break - return text.strip() + messages = [{"role": "system", "content": system}, + {"role": "user", "content": user}] + opts = {**(options or {}), "temperature": 0.2, "num_predict": num_predict} + # think=False : un appel utilitaire (plan, critique) ne doit jamais + # « réfléchir » — sur un modèle thinking, la pensée dévore le budget et + # multiplie la latence. Repli sans le paramètre si le modèle le refuse. + think: bool | None = False + while True: + text = "" + try: + async for chunk in ollama.chat( + model, messages, options=opts, think=think, + keep_alive=keep_alive, stream=True, + ): + text += chunk.get("message", {}).get("content", "") + if chunk.get("done"): + break + return text.strip() + except OllamaError as exc: + if think is False and "think" in str(exc).lower(): + think = None + continue + raise async def make_plan( diff --git a/backend/app/memory.py b/backend/app/memory.py index a6d3ea0..ae01e96 100644 --- a/backend/app/memory.py +++ b/backend/app/memory.py @@ -14,7 +14,7 @@ import logging import httpx from . import db -from .ollama_client import ollama +from .ollama_client import OllamaError, ollama logger = logging.getLogger(__name__) @@ -79,20 +79,30 @@ async def maybe_summarize( if previous: transcript = f"[résumé existant] {previous}\n{transcript}" - text = "" - async for chunk in ollama.chat( - model, - [ - {"role": "system", "content": _SUMMARY_PROMPT}, - {"role": "user", "content": transcript}, - ], - options={**(options or {}), "temperature": 0.2, "num_predict": 350}, - keep_alive=keep_alive, - stream=True, - ): - text += chunk.get("message", {}).get("content", "") - if chunk.get("done"): + messages = [ + {"role": "system", "content": _SUMMARY_PROMPT}, + {"role": "user", "content": transcript}, + ] + opts = {**(options or {}), "temperature": 0.2, "num_predict": 350} + # think=False : le résumé d'arrière-plan ne doit pas « réfléchir » + # (latence ×3 sur un modèle thinking). Repli si paramètre refusé. + think: bool | None = False + while True: + text = "" + try: + async for chunk in ollama.chat( + model, messages, options=opts, think=think, + keep_alive=keep_alive, stream=True, + ): + text += chunk.get("message", {}).get("content", "") + if chunk.get("done"): + break break + except OllamaError as exc: + if think is False and "think" in str(exc).lower(): + think = None + continue + raise if text.strip(): db.set_session_summary(sid, text.strip()) diff --git a/backend/app/routes/chat.py b/backend/app/routes/chat.py index b61589f..c5a834f 100644 --- a/backend/app/routes/chat.py +++ b/backend/app/routes/chat.py @@ -324,7 +324,10 @@ async def chat(req: ChatRequest) -> StreamingResponse: from ..mcp_client import manager as mcp_manager - memories, plan, code_model, mcp_tools = await asyncio.gather( + # La préparation peut déclencher un CHARGEMENT de modèle (plan, embed) + # qui dure plusieurs minutes sur un gros modèle. Sans battement de + # cœur, le silence SSE fait couper la connexion par le reverse proxy. + prep = asyncio.ensure_future(asyncio.gather( rag.recall(req.session_id, req.content, embed_model=cfg.get("embed_model")) if want_rag else asyncio.sleep(0, result=[]), enhance.make_plan(model, req.content, options=run_opts, keep_alive=keep) @@ -332,7 +335,15 @@ async def chat(req: ChatRequest) -> StreamingResponse: coder.pick_code_model(model, cfg.get("code_model")) if use_code else asyncio.sleep(0, result=model), mcp_manager.tool_definitions(), - ) + )) + while True: + try: + memories, plan, code_model, mcp_tools = await asyncio.wait_for( + asyncio.shield(prep), timeout=10.0 + ) + break + except TimeoutError: + yield _sse("ping", {"status": "waiting"}) # Pannes MCP éventuelles : notice non bloquante dans le fil. for mcp_notice in mcp_manager.notices():