mirror of
https://github.com/R0m1k3/Loki.git
synced 2026-10-11 17:26:57 +02:00
fix(chat): heartbeat SSE pendant la préparation + think=False sur les appels auxiliaires
Le déplacement de la préparation (plan, recall RAG) dans le flux SSE laissait un silence total pendant le chargement d'un gros modèle : le reverse proxy coupait la connexion (« Connexion interrompue pendant le chargement du modèle »). Avec la réflexion active, l'appel de plan laissait en plus le modèle penser — silence encore plus long. - ping SSE toutes les 10 s pendant la préparation du contexte - plan / auto-critique / résumé : think=False (repli automatique si le modèle refuse le paramètre) — un appel utilitaire ne réfléchit jamais Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
1 parent
da76b8192d
commit
a1380d0bf8
3 files changed
+48
-17
No files matched your search
+17
-7
@@ -12,7 +12,7 @@ import re
|
||||
|
||||
import httpx
|
||||
|
||||
from .ollama_client import ollama
|
||||
from .ollama_client import OllamaError, ollama
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -61,19 +61,29 @@ async def _ask(
|
||||
principal : un appel avec des options divergentes force Ollama à recharger
|
||||
le modèle en plein message.
|
||||
"""
|
||||
messages = [{"role": "system", "content": system},
|
||||
{"role": "user", "content": user}]
|
||||
opts = {**(options or {}), "temperature": 0.2, "num_predict": num_predict}
|
||||
# think=False : un appel utilitaire (plan, critique) ne doit jamais
|
||||
# « réfléchir » — sur un modèle thinking, la pensée dévore le budget et
|
||||
# multiplie la latence. Repli sans le paramètre si le modèle le refuse.
|
||||
think: bool | None = False
|
||||
while True:
|
||||
text = ""
|
||||
try:
|
||||
async for chunk in ollama.chat(
|
||||
model,
|
||||
[{"role": "system", "content": system},
|
||||
{"role": "user", "content": user}],
|
||||
options={**(options or {}), "temperature": 0.2, "num_predict": num_predict},
|
||||
keep_alive=keep_alive,
|
||||
stream=True,
|
||||
model, messages, options=opts, think=think,
|
||||
keep_alive=keep_alive, stream=True,
|
||||
):
|
||||
text += chunk.get("message", {}).get("content", "")
|
||||
if chunk.get("done"):
|
||||
break
|
||||
return text.strip()
|
||||
except OllamaError as exc:
|
||||
if think is False and "think" in str(exc).lower():
|
||||
think = None
|
||||
continue
|
||||
raise
|
||||
|
||||
|
||||
async def make_plan(
|
||||
|
||||
+19
-9
@@ -14,7 +14,7 @@ import logging
|
||||
import httpx
|
||||
|
||||
from . import db
|
||||
from .ollama_client import ollama
|
||||
from .ollama_client import OllamaError, ollama
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -79,20 +79,30 @@ async def maybe_summarize(
|
||||
if previous:
|
||||
transcript = f"[résumé existant] {previous}\n{transcript}"
|
||||
|
||||
text = ""
|
||||
async for chunk in ollama.chat(
|
||||
model,
|
||||
[
|
||||
messages = [
|
||||
{"role": "system", "content": _SUMMARY_PROMPT},
|
||||
{"role": "user", "content": transcript},
|
||||
],
|
||||
options={**(options or {}), "temperature": 0.2, "num_predict": 350},
|
||||
keep_alive=keep_alive,
|
||||
stream=True,
|
||||
]
|
||||
opts = {**(options or {}), "temperature": 0.2, "num_predict": 350}
|
||||
# think=False : le résumé d'arrière-plan ne doit pas « réfléchir »
|
||||
# (latence ×3 sur un modèle thinking). Repli si paramètre refusé.
|
||||
think: bool | None = False
|
||||
while True:
|
||||
text = ""
|
||||
try:
|
||||
async for chunk in ollama.chat(
|
||||
model, messages, options=opts, think=think,
|
||||
keep_alive=keep_alive, stream=True,
|
||||
):
|
||||
text += chunk.get("message", {}).get("content", "")
|
||||
if chunk.get("done"):
|
||||
break
|
||||
break
|
||||
except OllamaError as exc:
|
||||
if think is False and "think" in str(exc).lower():
|
||||
think = None
|
||||
continue
|
||||
raise
|
||||
|
||||
if text.strip():
|
||||
db.set_session_summary(sid, text.strip())
|
||||
|
||||
@@ -324,7 +324,10 @@ async def chat(req: ChatRequest) -> StreamingResponse:
|
||||
|
||||
from ..mcp_client import manager as mcp_manager
|
||||
|
||||
memories, plan, code_model, mcp_tools = await asyncio.gather(
|
||||
# La préparation peut déclencher un CHARGEMENT de modèle (plan, embed)
|
||||
# qui dure plusieurs minutes sur un gros modèle. Sans battement de
|
||||
# cœur, le silence SSE fait couper la connexion par le reverse proxy.
|
||||
prep = asyncio.ensure_future(asyncio.gather(
|
||||
rag.recall(req.session_id, req.content, embed_model=cfg.get("embed_model"))
|
||||
if want_rag else asyncio.sleep(0, result=[]),
|
||||
enhance.make_plan(model, req.content, options=run_opts, keep_alive=keep)
|
||||
@@ -332,7 +335,15 @@ async def chat(req: ChatRequest) -> StreamingResponse:
|
||||
coder.pick_code_model(model, cfg.get("code_model"))
|
||||
if use_code else asyncio.sleep(0, result=model),
|
||||
mcp_manager.tool_definitions(),
|
||||
))
|
||||
while True:
|
||||
try:
|
||||
memories, plan, code_model, mcp_tools = await asyncio.wait_for(
|
||||
asyncio.shield(prep), timeout=10.0
|
||||
)
|
||||
break
|
||||
except TimeoutError:
|
||||
yield _sse("ping", {"status": "waiting"})
|
||||
|
||||
# Pannes MCP éventuelles : notice non bloquante dans le fil.
|
||||
for mcp_notice in mcp_manager.notices():
|
||||
|
||||
Reference in new issue
Block a user